Files
llm/scripts/profile_1c_tabular_templates.py
2026-08-14 09:40:51 +03:00

392 lines
17 KiB
Python

from __future__ import annotations
import argparse
import json
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path
from typing import Any
from urllib import request
def rpc(adapter_url: str, method: str, payload: dict[str, Any], *, timeout: int = 240) -> dict[str, Any]:
body = json.dumps({"method": method, "payload": payload}, ensure_ascii=False).encode("utf-8")
req = request.Request(
f"{adapter_url.rstrip('/')}/rpc",
data=body,
headers={"Content-Type": "application/json; charset=utf-8"},
method="POST",
)
with request.urlopen(req, timeout=timeout) as resp:
return json.loads(resp.read().decode("utf-8", errors="replace"))
def read_inventory(path: Path) -> dict[str, Any]:
return json.loads(path.read_text(encoding="utf-8"))
def structure_summary(structure: dict[str, Any], sample_limit: int) -> dict[str, Any]:
cells = structure.get("cells") or []
parameters = structure.get("parameters") or []
cell_parameters = structure.get("cell_parameters") or []
cell_text_identifiers = structure.get("cell_text_identifiers") or []
cell_coordinate_hints = structure.get("cell_coordinate_hints") or []
named_areas = structure.get("named_areas") or []
named_ranges = structure.get("named_range_candidates") or []
styles = structure.get("cell_style_candidates") or []
column_widths = structure.get("column_widths") or []
format_table = structure.get("format_table") or []
font_table = structure.get("font_table") or []
cell_format_links = structure.get("cell_format_links") or []
merged_ranges = structure.get("merged_ranges") or []
merged_range_candidates = structure.get("merged_range_candidates") or []
merge_record_block_candidates = structure.get("merge_record_block_candidates") or []
merge_count_hints = structure.get("merge_count_hints") or []
style_coordinate_hints = [
item
for item in styles
if isinstance(item, dict) and isinstance(item.get("coordinate_hints"), dict)
]
diagnostics = structure.get("diagnostics") or []
top_level = structure.get("moxel_record_diagnostics") or {}
record_summary = top_level.get("top_level_record_summary") if isinstance(top_level, dict) else {}
counts = {
**(structure.get("counts") or {}),
"cells": int((structure.get("counts") or {}).get("cells") or len(cells)),
"parameters": int((structure.get("counts") or {}).get("parameters") or len(parameters)),
"cell_parameters": int((structure.get("counts") or {}).get("cell_parameters") or len(cell_parameters)),
"cell_text_identifiers": int((structure.get("counts") or {}).get("cell_text_identifiers") or len(cell_text_identifiers)),
"cell_coordinate_hints": int((structure.get("counts") or {}).get("cell_coordinate_hints") or len(cell_coordinate_hints)),
"named_areas": int((structure.get("counts") or {}).get("named_areas") or len(named_areas)),
"named_range_candidates": int((structure.get("counts") or {}).get("named_range_candidates") or len(named_ranges)),
"cell_style_candidates": int((structure.get("counts") or {}).get("cell_style_candidates") or len(styles)),
"cell_style_coordinate_hints": int((structure.get("counts") or {}).get("cell_style_coordinate_hints") or len(style_coordinate_hints)),
"column_widths": int((structure.get("counts") or {}).get("column_widths") or len(column_widths)),
"format_table": int((structure.get("counts") or {}).get("format_table") or len(format_table)),
"font_table": int((structure.get("counts") or {}).get("font_table") or len(font_table)),
"cell_format_links": int((structure.get("counts") or {}).get("cell_format_links") or len(cell_format_links)),
"merged_ranges": int((structure.get("counts") or {}).get("merged_ranges") or len(merged_ranges)),
"merged_range_candidates": int((structure.get("counts") or {}).get("merged_range_candidates") or len(merged_range_candidates)),
"merge_record_block_candidates": int((structure.get("counts") or {}).get("merge_record_block_candidates") or len(merge_record_block_candidates)),
"merge_count_hints": int((structure.get("counts") or {}).get("merge_count_hints") or len(merge_count_hints)),
}
return {
"format": structure.get("format"),
"dimensions": structure.get("dimensions"),
"capacity_dimensions": structure.get("capacity_dimensions"),
"used_dimensions": structure.get("used_dimensions"),
"format_dimensions": structure.get("format_dimensions"),
"capabilities": structure.get("capabilities") or {},
"counts": counts,
"sample_texts": [str(item.get("text") or "") for item in cells[:sample_limit] if isinstance(item, dict) and item.get("text")],
"sample_parameters": [str(item.get("name") or "") for item in parameters[:sample_limit] if isinstance(item, dict) and item.get("name")],
"sample_cell_parameters": [
{
"name": item.get("name"),
"row": item.get("row"),
"column": item.get("column"),
"cell_text": item.get("cell_text"),
}
for item in cell_parameters[:sample_limit]
if isinstance(item, dict)
],
"sample_text_identifiers": [
{
"name": item.get("name"),
"row": item.get("row"),
"column": item.get("column"),
"parameter": item.get("parameter"),
}
for item in cell_text_identifiers[:sample_limit]
if isinstance(item, dict)
],
"sample_cell_coordinate_hints": [
{
"text": item.get("text"),
"one_based": item.get("one_based"),
"confidence": item.get("confidence"),
"evidence": item.get("evidence"),
}
for item in cell_coordinate_hints[:sample_limit]
if isinstance(item, dict)
],
"sample_named_areas": [item.get("name") for item in named_areas[:sample_limit] if isinstance(item, dict) and item.get("name")],
"sample_named_ranges": [item.get("name") for item in named_ranges[:sample_limit] if isinstance(item, dict) and item.get("name")],
"sample_style_texts": [str(item.get("text") or "") for item in styles[:sample_limit] if isinstance(item, dict) and item.get("text")],
"sample_style_coordinate_hints": [
{
"text": item.get("text"),
"tree_position": item.get("tree_position"),
"coordinate_hints": item.get("coordinate_hints"),
}
for item in style_coordinate_hints[:sample_limit]
if isinstance(item, dict)
],
"sample_column_widths": [
{
"column": item.get("column"),
"width": item.get("width"),
}
for item in column_widths[:sample_limit]
if isinstance(item, dict)
],
"sample_format_table": [
{
"format_index": item.get("format_index"),
"font_index": item.get("font_index"),
"width": item.get("width"),
"horizontal_alignment": item.get("horizontal_alignment"),
"vertical_alignment": item.get("vertical_alignment"),
"text_placement": item.get("text_placement"),
}
for item in format_table[:sample_limit]
if isinstance(item, dict)
],
"sample_font_table": [
{
"font_index": item.get("font_index"),
"face_name": item.get("face_name"),
"height": item.get("height"),
"weight": item.get("weight"),
"bold": item.get("bold"),
}
for item in font_table[:sample_limit]
if isinstance(item, dict)
],
"sample_cell_format_links": [
{
"row": item.get("row"),
"column": item.get("column"),
"format_index": item.get("format_index"),
"format": item.get("format"),
"text": item.get("text"),
}
for item in cell_format_links[:sample_limit]
if isinstance(item, dict)
],
"cell_format_link_stats": structure.get("cell_format_link_stats") or {},
"sample_merged_ranges": [
{
"range": item.get("range"),
"source": item.get("source"),
}
for item in merged_ranges[:sample_limit]
if isinstance(item, dict)
],
"sample_merged_range_candidates": [
{
"name": item.get("name"),
"range": item.get("range"),
"confidence": item.get("confidence"),
}
for item in merged_range_candidates[:sample_limit]
if isinstance(item, dict)
],
"sample_merge_record_block_candidates": [
{
"count": item.get("count"),
"tree_position": item.get("tree_position"),
"record_window": item.get("record_window"),
"confidence": item.get("confidence"),
"evidence": item.get("evidence"),
"sample_records": item.get("sample_records"),
}
for item in merge_record_block_candidates[:sample_limit]
if isinstance(item, dict)
],
"sample_merge_count_hints": [
{
"count": item.get("count"),
"tree_position": item.get("tree_position"),
"confidence": item.get("confidence"),
"evidence": item.get("evidence"),
}
for item in merge_count_hints[:sample_limit]
if isinstance(item, dict)
],
"diagnostics": diagnostics[:sample_limit] if isinstance(diagnostics, list) else [],
"top_level_record_summary": record_summary or None,
}
def profile_template(
adapter_url: str,
base_id: str,
owner: dict[str, Any],
template: dict[str, Any],
timeout_seconds: int,
sample_limit: int,
refresh_cache: bool,
) -> dict[str, Any]:
result = rpc(
adapter_url,
"templates.read",
{
"base_id": base_id,
"guid": owner.get("guid"),
"kind": owner.get("kind"),
"name": owner.get("name"),
"template": template.get("name"),
"sections": "summary,merged_ranges,merges,coordinate_hints,cells,parameters,named_areas,formats,diagnostics",
"refresh_cache": refresh_cache,
"max_cells": sample_limit,
"max_merged": sample_limit,
"max_parameters": sample_limit,
"max_named_areas": sample_limit,
"timeout_seconds": timeout_seconds,
},
timeout=max(240, timeout_seconds + 60),
)
templates = result.get("templates") or []
if not templates:
return {
"owner": owner,
"template": {"name": template.get("name"), "guid": template.get("guid"), "synonym": template.get("synonym")},
"status": "not_found",
}
details = templates[0] or {}
payload_parts = [part for part in (details.get("parts") or []) if isinstance(part, dict) and (part.get("features") or {}).get("tabular_document")]
part = payload_parts[0] if payload_parts else {}
structure = (part.get("structure") or {}) if isinstance(part, dict) else {}
return {
"owner": owner,
"template": {
"guid": details.get("guid"),
"name": details.get("name"),
"synonym": details.get("synonym"),
"format": details.get("format"),
"platform_type": details.get("platform_type"),
},
"status": details.get("status"),
"features": details.get("features") or {},
"template_type_candidates": details.get("template_type_candidates") or [],
"profile": structure_summary(structure, sample_limit),
}
def render_markdown(payload: dict[str, Any]) -> str:
def dimension_text(value: dict[str, Any]) -> str:
if not isinstance(value, dict) or not value:
return ""
rows = value.get("rows") if value.get("rows") is not None else "-"
columns = value.get("columns") if value.get("columns") is not None else "-"
return f"{rows}x{columns}"
lines: list[str] = []
lines.append("# 1C tabular template profiles")
lines.append("")
lines.append(f"- Base: `{payload.get('base_id')}`")
lines.append(f"- Templates profiled: `{payload.get('counts', {}).get('templates_profiled')}`")
lines.append(f"- Objects covered: `{payload.get('counts', {}).get('objects_with_templates')}`")
lines.append("")
lines.append("| Owner | Template | Capacity | Used | Cells | Cell params | Text ids | Coord hints | Style hints | Merges | Merge blocks | Merge count hints | Formats | Named areas |")
lines.append("| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |")
for item in payload.get("items") or []:
owner = item.get("owner") or {}
template = item.get("template") or {}
profile = item.get("profile") or {}
dimensions = profile.get("capacity_dimensions") or profile.get("dimensions") or {}
used_dimensions = profile.get("used_dimensions") or {}
format_dimensions = profile.get("format_dimensions") or {}
counts = profile.get("counts") or {}
dim = dimension_text(dimensions)
used = dimension_text(used_dimensions)
formats = (
dimension_text(format_dimensions)
if format_dimensions
else counts.get("cell_style_candidates")
)
lines.append(
f"| `{owner.get('kind')}.{owner.get('name')}` | `{template.get('name')}` | `{dim}` | `{used}` | "
f"`{counts.get('cells')}` | `{counts.get('cell_parameters')}` | `{counts.get('cell_text_identifiers')}` | "
f"`{counts.get('cell_coordinate_hints')}` | "
f"`{counts.get('cell_style_coordinate_hints')}` | "
f"`{counts.get('merged_ranges')}` | `{counts.get('merge_record_block_candidates')}` | "
f"`{counts.get('merge_count_hints')}` | `{formats}` | "
f"`{counts.get('named_areas')}` |"
)
return "\n".join(lines) + "\n"
def main() -> int:
parser = argparse.ArgumentParser(description="Build compact structural profiles for all inventoried tabular 1C templates.")
parser.add_argument(
"--inventory-json",
default=str(Path("Z:/codex/LLM/reports/1c-template-baselines/upo_test_configuration_tabular_templates.json")),
)
parser.add_argument("--adapter-url", default="http://docker.cin.su:8011")
parser.add_argument("--base-id", default="upo_test")
parser.add_argument("--workers", type=int, default=8)
parser.add_argument("--timeout-seconds", type=int, default=120)
parser.add_argument("--sample-limit", type=int, default=10)
parser.add_argument("--refresh-cache", action="store_true")
parser.add_argument(
"--output-json",
default=str(Path("Z:/codex/LLM/reports/1c-template-baselines/upo_test_tabular_template_profiles.json")),
)
parser.add_argument(
"--output-markdown",
default=str(Path("Z:/codex/LLM/reports/1c-template-baselines/upo_test_tabular_template_profiles.md")),
)
args = parser.parse_args()
inventory = read_inventory(Path(args.inventory_json))
work_items: list[tuple[dict[str, Any], dict[str, Any]]] = []
for item in inventory.get("items") or []:
owner = item.get("object") or {}
for template in item.get("templates") or []:
work_items.append((owner, template))
results: list[dict[str, Any]] = []
with ThreadPoolExecutor(max_workers=max(1, int(args.workers or 1))) as pool:
future_map = {
pool.submit(
profile_template,
args.adapter_url,
args.base_id,
owner,
template,
int(args.timeout_seconds or 120),
int(args.sample_limit or 10),
bool(args.refresh_cache),
): (owner, template)
for owner, template in work_items
}
for future in as_completed(future_map):
results.append(future.result())
results.sort(key=lambda item: ((item.get("owner") or {}).get("name") or "", (item.get("template") or {}).get("name") or ""))
payload = {
"schema": "codex_1c_tabular_template_profiles.v1",
"inventory_json": args.inventory_json,
"adapter_url": args.adapter_url,
"base_id": args.base_id,
"items": results,
"counts": {
"templates_profiled": len(results),
"objects_with_templates": len({((item.get("owner") or {}).get("guid") or (item.get("owner") or {}).get("name")) for item in results}),
},
}
json_path = Path(args.output_json)
md_path = Path(args.output_markdown)
json_path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
md_path.write_text(render_markdown(payload), encoding="utf-8")
print(
json.dumps(
{
"status": "ok",
"json": str(json_path),
"markdown": str(md_path),
"counts": payload["counts"],
},
ensure_ascii=False,
indent=2,
)
)
return 0
if __name__ == "__main__":
raise SystemExit(main())