Initial SQL-only 1C adapter baseline

This commit is contained in:
2026-07-22 03:03:47 +03:00
commit e2503b77e7
545 changed files with 184711 additions and 0 deletions
+367
View File
@@ -0,0 +1,367 @@
from __future__ import annotations
import argparse
import itertools
import json
import re
import xml.etree.ElementTree as ET
from pathlib import Path
from typing import Any
DEFAULT_XML_ROOT = Path(r"Z:\codex\1C\XML\UPO\Структура базы 1с\Конфигурация")
PLACEHOLDER_RE = re.compile(r"\[([A-Za-zА-Яа-яЁё_][A-Za-zА-Яа-яЁё0-9_.]*)\]")
STYLE_TAGS = {
"format",
"formatIndex",
"f",
"width",
"height",
"horizontalAlignment",
"verticalAlignment",
"border",
"font",
"textColor",
"backgroundColor",
}
def namespace_uri(value: str) -> str | None:
if value.startswith("{") and "}" in value:
return value[1:].split("}", 1)[0]
return None
def local_name(value: str) -> str:
return value.rsplit("}", 1)[-1] if "}" in value else value
def xml_kind(root: ET.Element) -> str:
name = local_name(root.tag)
namespace = namespace_uri(root.tag) or ""
if name == "document" and "data/spreadsheet" in namespace:
return "tabular_document"
if name == "DataCompositionSchema":
return "data_composition_schema"
return name
def text_value(node: ET.Element | None) -> str | None:
if node is None or node.text is None:
return None
value = node.text.strip()
return value or None
def as_int(value: Any) -> int | None:
try:
return int(str(value).strip())
except (TypeError, ValueError):
return None
def children(node: ET.Element, name: str | None = None) -> list[ET.Element]:
items = list(node)
if name is None:
return items
return [item for item in items if local_name(item.tag) == name]
def descendants(node: ET.Element, name: str | None = None) -> list[ET.Element]:
result = []
for item in node.iter():
if item is node:
continue
if name is None or local_name(item.tag) == name:
result.append(item)
return result
def first_child_text(node: ET.Element, name: str) -> str | None:
for item in children(node, name):
value = text_value(item)
if value is not None:
return value
return None
def direct_columns_size(root: ET.Element) -> int | None:
for columns in children(root, "columns"):
size = as_int(first_child_text(columns, "size"))
if size is not None:
return size
return None
def direct_height(root: ET.Element) -> int | None:
for name in ("height", "vgRows"):
value = as_int(first_child_text(root, name))
if value is not None:
return value
return None
def cell_texts(cell: ET.Element) -> list[str]:
values = []
for item in descendants(cell):
if local_name(item.tag) in {"content", "parameter"}:
value = text_value(item)
if value:
values.append(value)
return values
def row_cells(row: ET.Element) -> list[dict[str, Any]]:
result = []
current_column = 0
for wrapper in children(row, "c"):
explicit_index = as_int(first_child_text(wrapper, "i"))
if explicit_index is not None:
current_column = explicit_index
payload = next((item for item in children(wrapper, "c")), wrapper)
texts = cell_texts(payload)
parameter = text_value(next((item for item in descendants(payload, "parameter")), None))
format_index = as_int(first_child_text(payload, "f")) or as_int(first_child_text(payload, "formatIndex"))
if texts or parameter or format_index is not None:
result.append(
{
"column": current_column + 1,
"zero_based": {"column": current_column},
"formatIndex": format_index,
"texts": texts,
**({"parameter": parameter} if parameter else {}),
}
)
current_column += 1
return result
def merge_ranges(root: ET.Element, *, limit: int = 200) -> list[dict[str, Any]]:
result = []
def append_range(row: int | None, column: int | None, width: int | None, height: int | None) -> bool:
if row is None or column is None:
return False
width = width or 1
height = height or 1
result.append(
{
"row": row + 1,
"column": column + 1,
"width": width,
"height": height,
"range": {
"one_based": {
"top": row + 1,
"left": column + 1,
"bottom": row + height,
"right": column + width,
},
"zero_based": {
"top": row,
"left": column,
"bottom": row + height - 1,
"right": column + width - 1,
},
},
"source": "xml_template_merge",
}
)
return len(result) >= limit
for merge in descendants(root, "merge"):
scalar_values = [(local_name(item.tag), as_int(text_value(item))) for item in children(merge)]
index = 0
while index < len(scalar_values):
if scalar_values[index][0] != "r":
index += 1
continue
row = scalar_values[index][1]
column = None
width = None
height = None
cursor = index + 1
while cursor < len(scalar_values) and scalar_values[cursor][0] != "r":
name, value = scalar_values[cursor]
if name == "c":
column = value
elif name == "w":
width = value
elif name == "h":
height = value
cursor += 1
if append_range(row, column, width, height):
return result
index = cursor
for item in descendants(merge, "r"):
row = as_int(first_child_text(item, "r"))
column = as_int(first_child_text(item, "c"))
width = as_int(first_child_text(item, "w")) or 1
height = as_int(first_child_text(item, "h")) or 1
if append_range(row, column, width, height):
return result
return result
def profile_template(path: Path, root_dir: Path) -> dict[str, Any]:
xml_root = ET.parse(path).getroot()
rows = []
max_row = 0
max_column = 0
parameter_names: list[str] = []
text_values: list[str] = []
placeholder_names: list[str] = []
for rows_item in descendants(xml_root, "rowsItem"):
row_index = as_int(first_child_text(rows_item, "index"))
row = next((item for item in children(rows_item, "row")), None)
if row_index is None or row is None:
continue
cells = row_cells(row)
if cells:
max_row = max(max_row, row_index + 1)
for cell in cells:
max_column = max(max_column, int(cell.get("column") or 0))
for value in cell.get("texts") or []:
text_values.append(value)
for match in PLACEHOLDER_RE.finditer(value):
placeholder_names.append(match.group(1))
if cell.get("parameter"):
parameter_names.append(str(cell["parameter"]))
rows.append(
{
"index": row_index,
"row": row_index + 1,
"formatIndex": as_int(first_child_text(row, "formatIndex")),
"cells": cells[:50],
"cell_count": len(cells),
}
)
merges = merge_ranges(xml_root)
for item in merges:
one_based = (item.get("range") or {}).get("one_based") or {}
max_row = max(max_row, int(one_based.get("bottom") or 0))
max_column = max(max_column, int(one_based.get("right") or 0))
style_counts = {
name: sum(1 for item in descendants(xml_root, name) if text_value(item) is not None)
for name in sorted(STYLE_TAGS)
}
format_indexes = [
as_int(text_value(item))
for item in descendants(xml_root)
if local_name(item.tag) in {"formatIndex", "f"} and as_int(text_value(item)) is not None
]
capacity_rows = direct_height(xml_root)
capacity_columns = direct_columns_size(xml_root)
return {
"path": str(path),
"relative_path": str(path.relative_to(root_dir)) if path.is_relative_to(root_dir) else str(path),
"xml_kind": xml_kind(xml_root),
"xml_root": {"name": local_name(xml_root.tag), "namespace": namespace_uri(xml_root.tag)},
"capacity_dimensions": {"rows": capacity_rows, "columns": capacity_columns},
"used_dimensions": {"rows": max_row or None, "columns": max_column or None, "evidence": ["rowsItem", "cells"] + (["merge"] if merges else [])},
"counts": {
"rows": len(rows),
"cells": sum(int(row.get("cell_count") or 0) for row in rows),
"texts": len(text_values),
"parameters": len(set(parameter_names)),
"placeholders": len(set(placeholder_names)),
"merges": len(merges),
"format_indexes": len(format_indexes),
"distinct_format_indexes": len(set(format_indexes)),
},
"style_counts": style_counts,
"samples": {
"rows": rows[:20],
"texts": text_values[:50],
"parameters": sorted(set(parameter_names))[:50],
"placeholders": sorted(set(placeholder_names))[:50],
"merges": merges[:50],
},
}
def analyze(root: Path, *, limit: int | None = None) -> dict[str, Any]:
file_iter = root.rglob("Template.xml")
files = list(itertools.islice(file_iter, limit)) if limit is not None else list(file_iter)
templates = []
errors = []
for path in files:
try:
templates.append(profile_template(path, root))
except Exception as exc:
errors.append({"path": str(path), "error": str(exc)})
return {
"schema": "codex_1c_template_xml_profiles.v1",
"source": "xml_analysis_fixture_only",
"root": str(root),
"templates": templates,
"counts": {
"files": len(files),
"templates": len(templates),
"errors": len(errors),
"with_merges": sum(1 for item in templates if int((item.get("counts") or {}).get("merges") or 0) > 0),
"with_parameters": sum(1 for item in templates if int((item.get("counts") or {}).get("parameters") or 0) > 0),
"with_placeholders": sum(1 for item in templates if int((item.get("counts") or {}).get("placeholders") or 0) > 0),
"by_xml_kind": {
kind: sum(1 for item in templates if item.get("xml_kind") == kind)
for kind in sorted({str(item.get("xml_kind") or "unknown") for item in templates})
},
},
**({"errors": errors[:100]} if errors else {}),
}
def render_markdown(payload: dict[str, Any]) -> str:
def dimension_text(value: dict[str, Any]) -> str:
rows = value.get("rows") if value.get("rows") is not None else "-"
columns = value.get("columns") if value.get("columns") is not None else "-"
return f"{rows}x{columns}"
lines = ["# 1C Template XML Profiles", ""]
counts = payload.get("counts") or {}
lines.append(f"- Source: `{payload.get('source')}`")
lines.append(f"- Templates: `{counts.get('templates')}`")
lines.append(f"- With merges: `{counts.get('with_merges')}`")
lines.append(f"- With parameters: `{counts.get('with_parameters')}`")
lines.append(f"- Errors: `{counts.get('errors')}`")
lines.append("")
lines.append("| Template | XML kind | Capacity | Used | Cells | Texts | Params | Merges | Formats |")
lines.append("| --- | --- | --- | --- | ---: | ---: | ---: | ---: | ---: |")
for item in payload.get("templates") or []:
item_counts = item.get("counts") or {}
capacity = item.get("capacity_dimensions") or {}
used = item.get("used_dimensions") or {}
lines.append(
f"| `{item.get('relative_path')}` | "
f"`{item.get('xml_kind')}` | "
f"`{dimension_text(capacity)}` | "
f"`{dimension_text(used)}` | "
f"{item_counts.get('cells')} | {item_counts.get('texts')} | {item_counts.get('parameters')} | "
f"{item_counts.get('merges')} | {item_counts.get('distinct_format_indexes')} |"
)
lines.append("")
return "\n".join(lines)
def main() -> int:
parser = argparse.ArgumentParser(description="Profile 1C Template.xml spreadsheet exports as analysis fixtures for SQL MOXCEL decoding.")
parser.add_argument("--root", default=str(DEFAULT_XML_ROOT), help="XML export root. Use Конфигурация by default, not extensions.")
parser.add_argument("--limit", type=int, help="Optional max Template.xml files to scan.")
parser.add_argument("--output-json", default="reports/1c-template-baselines/xml-template-profiles.json")
parser.add_argument("--output-markdown", default="reports/1c-template-baselines/xml-template-profiles.md")
args = parser.parse_args()
root = Path(args.root).resolve()
payload = analyze(root, limit=args.limit)
json_path = Path(args.output_json)
md_path = Path(args.output_markdown)
json_path.parent.mkdir(parents=True, exist_ok=True)
md_path.parent.mkdir(parents=True, exist_ok=True)
json_path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
md_path.write_text(render_markdown(payload), encoding="utf-8")
print(json.dumps({"status": "ok", "json": str(json_path), "markdown": str(md_path), "counts": payload["counts"]}, ensure_ascii=False, indent=2))
return 0
if __name__ == "__main__":
raise SystemExit(main())