Initial SQL-only 1C adapter baseline
This commit is contained in:
@@ -0,0 +1,224 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Build a GUID index for a 1C XML configuration dump.
|
||||
|
||||
The index is intentionally mechanical: it records GUID occurrences in XML
|
||||
attributes and element text, plus the top metadata object declared by each
|
||||
file. It does not infer SQL table names or storage roles.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
GUID_RE = re.compile(
|
||||
r"[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}"
|
||||
)
|
||||
TOP_OBJECT_RE = re.compile(
|
||||
r"<(?P<kind>[A-Za-z][A-Za-z0-9]*)\s+[^>]*uuid=\"(?P<guid>"
|
||||
+ GUID_RE.pattern
|
||||
+ r")\"",
|
||||
re.S,
|
||||
)
|
||||
NAME_RE = re.compile(r"<Name>(?P<name>.*?)</Name>", re.S)
|
||||
SYNONYM_RE = re.compile(r"<(?:[A-Za-z0-9]+:)?content>(?P<content>.*?)</(?:[A-Za-z0-9]+:)?content>", re.S)
|
||||
|
||||
|
||||
def clean_xml_text(value: str) -> str:
|
||||
return repair_mojibake(
|
||||
value.replace(""", '"')
|
||||
.replace("'", "'")
|
||||
.replace("<", "<")
|
||||
.replace(">", ">")
|
||||
.replace("&", "&")
|
||||
.strip()
|
||||
)
|
||||
|
||||
|
||||
def repair_mojibake(value: str) -> str:
|
||||
if not value:
|
||||
return value
|
||||
for source_encoding in ("gbk", "cp1255", "cp1252", "latin1"):
|
||||
try:
|
||||
candidate = value.encode(source_encoding).decode("cp1251")
|
||||
except UnicodeError:
|
||||
continue
|
||||
candidate_cyrillic = sum(1 for char in candidate if "А" <= char <= "я" or char == "ё" or char == "Ё")
|
||||
value_cyrillic = sum(1 for char in value if "А" <= char <= "я" or char == "ё" or char == "Ё")
|
||||
if candidate_cyrillic > value_cyrillic:
|
||||
return repair_mojibake(candidate)
|
||||
cjk_count = sum(1 for char in value if "\u4e00" <= char <= "\u9fff")
|
||||
if cjk_count:
|
||||
try:
|
||||
candidate = value.encode("gbk").decode("cp1251")
|
||||
return repair_mojibake(candidate)
|
||||
except UnicodeError:
|
||||
pass
|
||||
cyrillic_count = sum(1 for char in value if "А" <= char <= "я" or char == "ё" or char == "Ё")
|
||||
suspicious_count = sum(1 for char in value if char in "ÐÑÂÃÄÅÆÇÈÉÊËÌÍÎÏÒÓÔÕÖרÙÚÛÜÝÞßàáâãäåæçèéêëìíîïòóôõö÷øùúûüýþÿ")
|
||||
if suspicious_count <= cyrillic_count:
|
||||
return value
|
||||
try:
|
||||
repaired = value.encode("latin1").decode("cp1251")
|
||||
except UnicodeError:
|
||||
return value
|
||||
repaired_cyrillic = sum(1 for char in repaired if "А" <= char <= "я" or char == "ё" or char == "Ё")
|
||||
return repaired if repaired_cyrillic > cyrillic_count else value
|
||||
|
||||
|
||||
def inspect_xml_file(path: Path, root_dir: Path, *, max_occurrences_per_file: int) -> dict[str, Any]:
|
||||
item: dict[str, Any] = {
|
||||
"path": repair_mojibake(str(path)),
|
||||
"relative_path": repair_mojibake(str(path.relative_to(root_dir))),
|
||||
"status": "ok",
|
||||
"root_tag": "",
|
||||
"top_object": None,
|
||||
"occurrences": [],
|
||||
}
|
||||
try:
|
||||
text = path.read_text(encoding="utf-8", errors="replace")
|
||||
except OSError as exc:
|
||||
item["status"] = "read_error"
|
||||
item["error"] = str(exc)
|
||||
return item
|
||||
|
||||
root_match = re.search(r"<(?P<tag>[A-Za-z][A-Za-z0-9]*)[\s>]", text)
|
||||
item["root_tag"] = root_match.group("tag") if root_match else ""
|
||||
|
||||
object_match = TOP_OBJECT_RE.search(text)
|
||||
if object_match:
|
||||
object_start = object_match.start()
|
||||
object_end = TOP_OBJECT_RE.search(text, object_match.end())
|
||||
object_fragment = text[object_start : object_end.start() if object_end else min(len(text), object_start + 500_000)]
|
||||
name_match = NAME_RE.search(object_fragment)
|
||||
synonym_match = SYNONYM_RE.search(object_fragment)
|
||||
top_object = {
|
||||
"guid": object_match.group("guid").lower(),
|
||||
"xml_kind": object_match.group("kind"),
|
||||
"name": clean_xml_text(name_match.group("name")) if name_match else "",
|
||||
"synonym": clean_xml_text(synonym_match.group("content")) if synonym_match else "",
|
||||
}
|
||||
else:
|
||||
top_object = None
|
||||
item["top_object"] = top_object
|
||||
|
||||
occurrences = []
|
||||
for match in GUID_RE.finditer(text):
|
||||
occurrences.append({"guid": match.group(0).lower(), "offset": match.start()})
|
||||
if len(occurrences) >= max_occurrences_per_file:
|
||||
item["occurrences"] = occurrences
|
||||
item["truncated"] = True
|
||||
return item
|
||||
item["occurrences"] = occurrences
|
||||
item["truncated"] = False
|
||||
return item
|
||||
|
||||
|
||||
def build_guid_map(files: list[dict[str, Any]], *, max_occurrences_per_guid: int) -> dict[str, Any]:
|
||||
guid_map: dict[str, Any] = {}
|
||||
for file_item in files:
|
||||
if file_item.get("status") != "ok":
|
||||
continue
|
||||
top_object = file_item.get("top_object")
|
||||
if top_object:
|
||||
guid = top_object["guid"]
|
||||
entry = guid_map.setdefault(guid, {"total_occurrences": 0, "top_objects": [], "occurrences": []})
|
||||
entry["top_objects"].append(
|
||||
{
|
||||
"path": file_item["path"],
|
||||
"relative_path": file_item["relative_path"],
|
||||
"xml_kind": top_object["xml_kind"],
|
||||
"name": top_object["name"],
|
||||
"synonym": top_object["synonym"],
|
||||
}
|
||||
)
|
||||
for occurrence in file_item.get("occurrences") or []:
|
||||
guid = occurrence["guid"]
|
||||
entry = guid_map.setdefault(guid, {"total_occurrences": 0, "top_objects": [], "occurrences": []})
|
||||
entry["total_occurrences"] += 1
|
||||
if len(entry["occurrences"]) < max_occurrences_per_guid:
|
||||
entry["occurrences"].append(
|
||||
{
|
||||
"path": file_item["path"],
|
||||
"relative_path": file_item["relative_path"],
|
||||
"offset": occurrence["offset"],
|
||||
}
|
||||
)
|
||||
return dict(sorted(guid_map.items()))
|
||||
|
||||
|
||||
def list_xml_paths(root: Path, max_relative_depth: int) -> list[Path]:
|
||||
if max_relative_depth <= 0:
|
||||
return sorted(root.rglob("*.xml"))
|
||||
result: list[Path] = []
|
||||
root_parts = len(root.parts)
|
||||
for current, dirs, files in os.walk(root):
|
||||
current_path = Path(current)
|
||||
relative_depth = len(current_path.parts) - root_parts
|
||||
if relative_depth >= max_relative_depth - 1:
|
||||
dirs[:] = []
|
||||
for file_name in files:
|
||||
if file_name.lower().endswith(".xml"):
|
||||
path = current_path / file_name
|
||||
if len(path.relative_to(root).parts) <= max_relative_depth:
|
||||
result.append(path)
|
||||
return sorted(result)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description="Build a GUID index for a 1C XML dump.")
|
||||
parser.add_argument("xml_root", type=Path)
|
||||
parser.add_argument("--output", type=Path, required=True)
|
||||
parser.add_argument("--max-occurrences-per-file", type=int, default=5000)
|
||||
parser.add_argument("--max-occurrences-per-guid", type=int, default=20)
|
||||
parser.add_argument(
|
||||
"--max-relative-depth",
|
||||
type=int,
|
||||
default=0,
|
||||
help="Only scan XML files whose relative path has at most this many parts; 0 scans all files.",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
xml_root = args.xml_root.resolve()
|
||||
paths = list_xml_paths(xml_root, args.max_relative_depth)
|
||||
files = [
|
||||
inspect_xml_file(path, xml_root, max_occurrences_per_file=args.max_occurrences_per_file)
|
||||
for path in paths
|
||||
]
|
||||
guid_map = build_guid_map(files, max_occurrences_per_guid=args.max_occurrences_per_guid)
|
||||
top_object_count = sum(1 for item in files if item.get("top_object"))
|
||||
parse_errors = sum(1 for item in files if item.get("status") != "ok")
|
||||
report = {
|
||||
"schema": "onec_xml_guid_index.v1",
|
||||
"xml_root": repair_mojibake(str(xml_root)),
|
||||
"file_count": len(files),
|
||||
"top_object_count": top_object_count,
|
||||
"parse_error_count": parse_errors,
|
||||
"guid_count": len(guid_map),
|
||||
"files": files,
|
||||
"guid_map": guid_map,
|
||||
}
|
||||
args.output.parent.mkdir(parents=True, exist_ok=True)
|
||||
args.output.write_text(json.dumps(report, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
"output": str(args.output),
|
||||
"files": len(files),
|
||||
"top_objects": top_object_count,
|
||||
"guids": len(guid_map),
|
||||
"parse_errors": parse_errors,
|
||||
},
|
||||
ensure_ascii=False,
|
||||
)
|
||||
)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user