Files
llm/scripts/analyze_1c_template_xml_profiles.py

368 lines
14 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
from __future__ import annotations
import argparse
import itertools
import json
import re
import xml.etree.ElementTree as ET
from pathlib import Path
from typing import Any
DEFAULT_XML_ROOT = Path(r"Z:\codex\1C\XML\UPO\Структура базы 1с\Конфигурация")
PLACEHOLDER_RE = re.compile(r"\[([A-Za-zА-Яа-яЁё_][A-Za-zА-Яа-яЁё0-9_.]*)\]")
STYLE_TAGS = {
"format",
"formatIndex",
"f",
"width",
"height",
"horizontalAlignment",
"verticalAlignment",
"border",
"font",
"textColor",
"backgroundColor",
}
def namespace_uri(value: str) -> str | None:
if value.startswith("{") and "}" in value:
return value[1:].split("}", 1)[0]
return None
def local_name(value: str) -> str:
return value.rsplit("}", 1)[-1] if "}" in value else value
def xml_kind(root: ET.Element) -> str:
name = local_name(root.tag)
namespace = namespace_uri(root.tag) or ""
if name == "document" and "data/spreadsheet" in namespace:
return "tabular_document"
if name == "DataCompositionSchema":
return "data_composition_schema"
return name
def text_value(node: ET.Element | None) -> str | None:
if node is None or node.text is None:
return None
value = node.text.strip()
return value or None
def as_int(value: Any) -> int | None:
try:
return int(str(value).strip())
except (TypeError, ValueError):
return None
def children(node: ET.Element, name: str | None = None) -> list[ET.Element]:
items = list(node)
if name is None:
return items
return [item for item in items if local_name(item.tag) == name]
def descendants(node: ET.Element, name: str | None = None) -> list[ET.Element]:
result = []
for item in node.iter():
if item is node:
continue
if name is None or local_name(item.tag) == name:
result.append(item)
return result
def first_child_text(node: ET.Element, name: str) -> str | None:
for item in children(node, name):
value = text_value(item)
if value is not None:
return value
return None
def direct_columns_size(root: ET.Element) -> int | None:
for columns in children(root, "columns"):
size = as_int(first_child_text(columns, "size"))
if size is not None:
return size
return None
def direct_height(root: ET.Element) -> int | None:
for name in ("height", "vgRows"):
value = as_int(first_child_text(root, name))
if value is not None:
return value
return None
def cell_texts(cell: ET.Element) -> list[str]:
values = []
for item in descendants(cell):
if local_name(item.tag) in {"content", "parameter"}:
value = text_value(item)
if value:
values.append(value)
return values
def row_cells(row: ET.Element) -> list[dict[str, Any]]:
result = []
current_column = 0
for wrapper in children(row, "c"):
explicit_index = as_int(first_child_text(wrapper, "i"))
if explicit_index is not None:
current_column = explicit_index
payload = next((item for item in children(wrapper, "c")), wrapper)
texts = cell_texts(payload)
parameter = text_value(next((item for item in descendants(payload, "parameter")), None))
format_index = as_int(first_child_text(payload, "f")) or as_int(first_child_text(payload, "formatIndex"))
if texts or parameter or format_index is not None:
result.append(
{
"column": current_column + 1,
"zero_based": {"column": current_column},
"formatIndex": format_index,
"texts": texts,
**({"parameter": parameter} if parameter else {}),
}
)
current_column += 1
return result
def merge_ranges(root: ET.Element, *, limit: int = 200) -> list[dict[str, Any]]:
result = []
def append_range(row: int | None, column: int | None, width: int | None, height: int | None) -> bool:
if row is None or column is None:
return False
width = width or 1
height = height or 1
result.append(
{
"row": row + 1,
"column": column + 1,
"width": width,
"height": height,
"range": {
"one_based": {
"top": row + 1,
"left": column + 1,
"bottom": row + height,
"right": column + width,
},
"zero_based": {
"top": row,
"left": column,
"bottom": row + height - 1,
"right": column + width - 1,
},
},
"source": "xml_template_merge",
}
)
return len(result) >= limit
for merge in descendants(root, "merge"):
scalar_values = [(local_name(item.tag), as_int(text_value(item))) for item in children(merge)]
index = 0
while index < len(scalar_values):
if scalar_values[index][0] != "r":
index += 1
continue
row = scalar_values[index][1]
column = None
width = None
height = None
cursor = index + 1
while cursor < len(scalar_values) and scalar_values[cursor][0] != "r":
name, value = scalar_values[cursor]
if name == "c":
column = value
elif name == "w":
width = value
elif name == "h":
height = value
cursor += 1
if append_range(row, column, width, height):
return result
index = cursor
for item in descendants(merge, "r"):
row = as_int(first_child_text(item, "r"))
column = as_int(first_child_text(item, "c"))
width = as_int(first_child_text(item, "w")) or 1
height = as_int(first_child_text(item, "h")) or 1
if append_range(row, column, width, height):
return result
return result
def profile_template(path: Path, root_dir: Path) -> dict[str, Any]:
xml_root = ET.parse(path).getroot()
rows = []
max_row = 0
max_column = 0
parameter_names: list[str] = []
text_values: list[str] = []
placeholder_names: list[str] = []
for rows_item in descendants(xml_root, "rowsItem"):
row_index = as_int(first_child_text(rows_item, "index"))
row = next((item for item in children(rows_item, "row")), None)
if row_index is None or row is None:
continue
cells = row_cells(row)
if cells:
max_row = max(max_row, row_index + 1)
for cell in cells:
max_column = max(max_column, int(cell.get("column") or 0))
for value in cell.get("texts") or []:
text_values.append(value)
for match in PLACEHOLDER_RE.finditer(value):
placeholder_names.append(match.group(1))
if cell.get("parameter"):
parameter_names.append(str(cell["parameter"]))
rows.append(
{
"index": row_index,
"row": row_index + 1,
"formatIndex": as_int(first_child_text(row, "formatIndex")),
"cells": cells[:50],
"cell_count": len(cells),
}
)
merges = merge_ranges(xml_root)
for item in merges:
one_based = (item.get("range") or {}).get("one_based") or {}
max_row = max(max_row, int(one_based.get("bottom") or 0))
max_column = max(max_column, int(one_based.get("right") or 0))
style_counts = {
name: sum(1 for item in descendants(xml_root, name) if text_value(item) is not None)
for name in sorted(STYLE_TAGS)
}
format_indexes = [
as_int(text_value(item))
for item in descendants(xml_root)
if local_name(item.tag) in {"formatIndex", "f"} and as_int(text_value(item)) is not None
]
capacity_rows = direct_height(xml_root)
capacity_columns = direct_columns_size(xml_root)
return {
"path": str(path),
"relative_path": str(path.relative_to(root_dir)) if path.is_relative_to(root_dir) else str(path),
"xml_kind": xml_kind(xml_root),
"xml_root": {"name": local_name(xml_root.tag), "namespace": namespace_uri(xml_root.tag)},
"capacity_dimensions": {"rows": capacity_rows, "columns": capacity_columns},
"used_dimensions": {"rows": max_row or None, "columns": max_column or None, "evidence": ["rowsItem", "cells"] + (["merge"] if merges else [])},
"counts": {
"rows": len(rows),
"cells": sum(int(row.get("cell_count") or 0) for row in rows),
"texts": len(text_values),
"parameters": len(set(parameter_names)),
"placeholders": len(set(placeholder_names)),
"merges": len(merges),
"format_indexes": len(format_indexes),
"distinct_format_indexes": len(set(format_indexes)),
},
"style_counts": style_counts,
"samples": {
"rows": rows[:20],
"texts": text_values[:50],
"parameters": sorted(set(parameter_names))[:50],
"placeholders": sorted(set(placeholder_names))[:50],
"merges": merges[:50],
},
}
def analyze(root: Path, *, limit: int | None = None) -> dict[str, Any]:
file_iter = root.rglob("Template.xml")
files = list(itertools.islice(file_iter, limit)) if limit is not None else list(file_iter)
templates = []
errors = []
for path in files:
try:
templates.append(profile_template(path, root))
except Exception as exc:
errors.append({"path": str(path), "error": str(exc)})
return {
"schema": "codex_1c_template_xml_profiles.v1",
"source": "xml_analysis_fixture_only",
"root": str(root),
"templates": templates,
"counts": {
"files": len(files),
"templates": len(templates),
"errors": len(errors),
"with_merges": sum(1 for item in templates if int((item.get("counts") or {}).get("merges") or 0) > 0),
"with_parameters": sum(1 for item in templates if int((item.get("counts") or {}).get("parameters") or 0) > 0),
"with_placeholders": sum(1 for item in templates if int((item.get("counts") or {}).get("placeholders") or 0) > 0),
"by_xml_kind": {
kind: sum(1 for item in templates if item.get("xml_kind") == kind)
for kind in sorted({str(item.get("xml_kind") or "unknown") for item in templates})
},
},
**({"errors": errors[:100]} if errors else {}),
}
def render_markdown(payload: dict[str, Any]) -> str:
def dimension_text(value: dict[str, Any]) -> str:
rows = value.get("rows") if value.get("rows") is not None else "-"
columns = value.get("columns") if value.get("columns") is not None else "-"
return f"{rows}x{columns}"
lines = ["# 1C Template XML Profiles", ""]
counts = payload.get("counts") or {}
lines.append(f"- Source: `{payload.get('source')}`")
lines.append(f"- Templates: `{counts.get('templates')}`")
lines.append(f"- With merges: `{counts.get('with_merges')}`")
lines.append(f"- With parameters: `{counts.get('with_parameters')}`")
lines.append(f"- Errors: `{counts.get('errors')}`")
lines.append("")
lines.append("| Template | XML kind | Capacity | Used | Cells | Texts | Params | Merges | Formats |")
lines.append("| --- | --- | --- | --- | ---: | ---: | ---: | ---: | ---: |")
for item in payload.get("templates") or []:
item_counts = item.get("counts") or {}
capacity = item.get("capacity_dimensions") or {}
used = item.get("used_dimensions") or {}
lines.append(
f"| `{item.get('relative_path')}` | "
f"`{item.get('xml_kind')}` | "
f"`{dimension_text(capacity)}` | "
f"`{dimension_text(used)}` | "
f"{item_counts.get('cells')} | {item_counts.get('texts')} | {item_counts.get('parameters')} | "
f"{item_counts.get('merges')} | {item_counts.get('distinct_format_indexes')} |"
)
lines.append("")
return "\n".join(lines)
def main() -> int:
parser = argparse.ArgumentParser(description="Profile 1C Template.xml spreadsheet exports as analysis fixtures for SQL MOXCEL decoding.")
parser.add_argument("--root", default=str(DEFAULT_XML_ROOT), help="XML export root. Use Конфигурация by default, not extensions.")
parser.add_argument("--limit", type=int, help="Optional max Template.xml files to scan.")
parser.add_argument("--output-json", default="reports/1c-template-baselines/xml-template-profiles.json")
parser.add_argument("--output-markdown", default="reports/1c-template-baselines/xml-template-profiles.md")
args = parser.parse_args()
root = Path(args.root).resolve()
payload = analyze(root, limit=args.limit)
json_path = Path(args.output_json)
md_path = Path(args.output_markdown)
json_path.parent.mkdir(parents=True, exist_ok=True)
md_path.parent.mkdir(parents=True, exist_ok=True)
json_path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
md_path.write_text(render_markdown(payload), encoding="utf-8")
print(json.dumps({"status": "ok", "json": str(json_path), "markdown": str(md_path), "counts": payload["counts"]}, ensure_ascii=False, indent=2))
return 0
if __name__ == "__main__":
raise SystemExit(main())