Initial SQL-only 1C adapter baseline

This commit is contained in:
2026-07-22 03:03:47 +03:00
commit e2503b77e7
545 changed files with 184711 additions and 0 deletions
+224
View File
@@ -0,0 +1,224 @@
#!/usr/bin/env python3
"""Build a GUID index for a 1C XML configuration dump.
The index is intentionally mechanical: it records GUID occurrences in XML
attributes and element text, plus the top metadata object declared by each
file. It does not infer SQL table names or storage roles.
"""
from __future__ import annotations
import argparse
import json
import os
import re
from pathlib import Path
from typing import Any
GUID_RE = re.compile(
r"[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}"
)
TOP_OBJECT_RE = re.compile(
r"<(?P<kind>[A-Za-z][A-Za-z0-9]*)\s+[^>]*uuid=\"(?P<guid>"
+ GUID_RE.pattern
+ r")\"",
re.S,
)
NAME_RE = re.compile(r"<Name>(?P<name>.*?)</Name>", re.S)
SYNONYM_RE = re.compile(r"<(?:[A-Za-z0-9]+:)?content>(?P<content>.*?)</(?:[A-Za-z0-9]+:)?content>", re.S)
def clean_xml_text(value: str) -> str:
return repair_mojibake(
value.replace("&quot;", '"')
.replace("&apos;", "'")
.replace("&lt;", "<")
.replace("&gt;", ">")
.replace("&amp;", "&")
.strip()
)
def repair_mojibake(value: str) -> str:
if not value:
return value
for source_encoding in ("gbk", "cp1255", "cp1252", "latin1"):
try:
candidate = value.encode(source_encoding).decode("cp1251")
except UnicodeError:
continue
candidate_cyrillic = sum(1 for char in candidate if "А" <= char <= "я" or char == "ё" or char == "Ё")
value_cyrillic = sum(1 for char in value if "А" <= char <= "я" or char == "ё" or char == "Ё")
if candidate_cyrillic > value_cyrillic:
return repair_mojibake(candidate)
cjk_count = sum(1 for char in value if "\u4e00" <= char <= "\u9fff")
if cjk_count:
try:
candidate = value.encode("gbk").decode("cp1251")
return repair_mojibake(candidate)
except UnicodeError:
pass
cyrillic_count = sum(1 for char in value if "А" <= char <= "я" or char == "ё" or char == "Ё")
suspicious_count = sum(1 for char in value if char in "ÐÑÂÃÄÅÆÇÈÉÊËÌÍÎÏÒÓÔÕÖרÙÚÛÜÝÞßàáâãäåæçèéêëìíîïòóôõö÷øùúûüýþÿ")
if suspicious_count <= cyrillic_count:
return value
try:
repaired = value.encode("latin1").decode("cp1251")
except UnicodeError:
return value
repaired_cyrillic = sum(1 for char in repaired if "А" <= char <= "я" or char == "ё" or char == "Ё")
return repaired if repaired_cyrillic > cyrillic_count else value
def inspect_xml_file(path: Path, root_dir: Path, *, max_occurrences_per_file: int) -> dict[str, Any]:
item: dict[str, Any] = {
"path": repair_mojibake(str(path)),
"relative_path": repair_mojibake(str(path.relative_to(root_dir))),
"status": "ok",
"root_tag": "",
"top_object": None,
"occurrences": [],
}
try:
text = path.read_text(encoding="utf-8", errors="replace")
except OSError as exc:
item["status"] = "read_error"
item["error"] = str(exc)
return item
root_match = re.search(r"<(?P<tag>[A-Za-z][A-Za-z0-9]*)[\s>]", text)
item["root_tag"] = root_match.group("tag") if root_match else ""
object_match = TOP_OBJECT_RE.search(text)
if object_match:
object_start = object_match.start()
object_end = TOP_OBJECT_RE.search(text, object_match.end())
object_fragment = text[object_start : object_end.start() if object_end else min(len(text), object_start + 500_000)]
name_match = NAME_RE.search(object_fragment)
synonym_match = SYNONYM_RE.search(object_fragment)
top_object = {
"guid": object_match.group("guid").lower(),
"xml_kind": object_match.group("kind"),
"name": clean_xml_text(name_match.group("name")) if name_match else "",
"synonym": clean_xml_text(synonym_match.group("content")) if synonym_match else "",
}
else:
top_object = None
item["top_object"] = top_object
occurrences = []
for match in GUID_RE.finditer(text):
occurrences.append({"guid": match.group(0).lower(), "offset": match.start()})
if len(occurrences) >= max_occurrences_per_file:
item["occurrences"] = occurrences
item["truncated"] = True
return item
item["occurrences"] = occurrences
item["truncated"] = False
return item
def build_guid_map(files: list[dict[str, Any]], *, max_occurrences_per_guid: int) -> dict[str, Any]:
guid_map: dict[str, Any] = {}
for file_item in files:
if file_item.get("status") != "ok":
continue
top_object = file_item.get("top_object")
if top_object:
guid = top_object["guid"]
entry = guid_map.setdefault(guid, {"total_occurrences": 0, "top_objects": [], "occurrences": []})
entry["top_objects"].append(
{
"path": file_item["path"],
"relative_path": file_item["relative_path"],
"xml_kind": top_object["xml_kind"],
"name": top_object["name"],
"synonym": top_object["synonym"],
}
)
for occurrence in file_item.get("occurrences") or []:
guid = occurrence["guid"]
entry = guid_map.setdefault(guid, {"total_occurrences": 0, "top_objects": [], "occurrences": []})
entry["total_occurrences"] += 1
if len(entry["occurrences"]) < max_occurrences_per_guid:
entry["occurrences"].append(
{
"path": file_item["path"],
"relative_path": file_item["relative_path"],
"offset": occurrence["offset"],
}
)
return dict(sorted(guid_map.items()))
def list_xml_paths(root: Path, max_relative_depth: int) -> list[Path]:
if max_relative_depth <= 0:
return sorted(root.rglob("*.xml"))
result: list[Path] = []
root_parts = len(root.parts)
for current, dirs, files in os.walk(root):
current_path = Path(current)
relative_depth = len(current_path.parts) - root_parts
if relative_depth >= max_relative_depth - 1:
dirs[:] = []
for file_name in files:
if file_name.lower().endswith(".xml"):
path = current_path / file_name
if len(path.relative_to(root).parts) <= max_relative_depth:
result.append(path)
return sorted(result)
def main() -> int:
parser = argparse.ArgumentParser(description="Build a GUID index for a 1C XML dump.")
parser.add_argument("xml_root", type=Path)
parser.add_argument("--output", type=Path, required=True)
parser.add_argument("--max-occurrences-per-file", type=int, default=5000)
parser.add_argument("--max-occurrences-per-guid", type=int, default=20)
parser.add_argument(
"--max-relative-depth",
type=int,
default=0,
help="Only scan XML files whose relative path has at most this many parts; 0 scans all files.",
)
args = parser.parse_args()
xml_root = args.xml_root.resolve()
paths = list_xml_paths(xml_root, args.max_relative_depth)
files = [
inspect_xml_file(path, xml_root, max_occurrences_per_file=args.max_occurrences_per_file)
for path in paths
]
guid_map = build_guid_map(files, max_occurrences_per_guid=args.max_occurrences_per_guid)
top_object_count = sum(1 for item in files if item.get("top_object"))
parse_errors = sum(1 for item in files if item.get("status") != "ok")
report = {
"schema": "onec_xml_guid_index.v1",
"xml_root": repair_mojibake(str(xml_root)),
"file_count": len(files),
"top_object_count": top_object_count,
"parse_error_count": parse_errors,
"guid_count": len(guid_map),
"files": files,
"guid_map": guid_map,
}
args.output.parent.mkdir(parents=True, exist_ok=True)
args.output.write_text(json.dumps(report, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
print(
json.dumps(
{
"output": str(args.output),
"files": len(files),
"top_objects": top_object_count,
"guids": len(guid_map),
"parse_errors": parse_errors,
},
ensure_ascii=False,
)
)
return 0
if __name__ == "__main__":
raise SystemExit(main())