from __future__ import annotations import argparse import json from pathlib import Path from typing import Any from urllib import request def rpc(adapter_url: str, method: str, payload: dict[str, Any]) -> dict[str, Any]: body = json.dumps({"method": method, "payload": payload}, ensure_ascii=False).encode("utf-8") req = request.Request( f"{adapter_url.rstrip('/')}/rpc", data=body, headers={"Content-Type": "application/json; charset=utf-8"}, method="POST", ) with request.urlopen(req, timeout=240) as resp: return json.loads(resp.read().decode("utf-8", errors="replace")) def latest_configcas_rows(adapter_url: str, base_id: str, limit: int) -> list[dict[str, Any]]: result = rpc( adapter_url, "query.run", { "base_id": base_id, "diagnostic": True, "query": ( f"SELECT TOP {int(limit)} FileName, DATALENGTH(BinaryData) AS Bytes, PartNo, Creation, Modified " "FROM ConfigCAS ORDER BY Modified DESC" ), "timeout_seconds": 120, }, ) return result.get("rows") or [] def read_structure(adapter_url: str, base_id: str, file_name: str, max_cells: int) -> dict[str, Any]: result = rpc( adapter_url, "templates.map", { "base_id": base_id, "table": "ConfigCAS", "file_name": file_name, "view": "summary", "sections": "cells,styles,named_areas,named_ranges,diagnostics", "max_cells": max_cells, "max_areas": 200, "timeout_seconds": 120, }, ) templates = result.get("templates") or [] if not templates: return {} item = templates[0] or {} structure = item.get("structure") or {} route = item.get("route") or {} return {"structure": structure, "route": route} def summarize_candidate(row: dict[str, Any], structure_bundle: dict[str, Any]) -> dict[str, Any]: structure = structure_bundle.get("structure") or {} route = structure_bundle.get("route") or {} counts = structure.get("counts") or {} named_areas = [item.get("name") for item in (structure.get("named_areas") or []) if isinstance(item, dict) and item.get("name")] named_ranges = [item.get("name") for item in (structure.get("named_range_candidates") or []) if isinstance(item, dict) and item.get("name")] cells = structure.get("cells") or [] styles = structure.get("cell_style_candidates") or [] sample_texts: list[str] = [] for item in cells: if not isinstance(item, dict): continue text = str(item.get("text") or "") if text and text not in sample_texts: sample_texts.append(text) if len(sample_texts) >= 12: break sample_style_texts: list[str] = [] for item in styles: if not isinstance(item, dict): continue text = str(item.get("text") or "") if text and text not in sample_style_texts: sample_style_texts.append(text) if len(sample_style_texts) >= 12: break return { "file_name": row.get("FileName"), "bytes": row.get("Bytes"), "modified": row.get("Modified"), "format": structure.get("format"), "dimensions": structure.get("dimensions"), "counts": counts, "capabilities": structure.get("capabilities") or {}, "named_areas": named_areas, "named_ranges": named_ranges, "sample_texts": sample_texts, "sample_style_texts": sample_style_texts, "route": route, } def is_interesting(summary: dict[str, Any]) -> bool: if str(summary.get("format") or "") == "MOXCEL": return True counts = summary.get("counts") or {} return any( int(counts.get(key) or 0) > 0 for key in ("cells", "cell_style_candidates", "named_areas", "named_range_candidates") ) def render_markdown(payload: dict[str, Any]) -> str: lines: list[str] = [] lines.append("# 1C template payload inventory") lines.append("") lines.append(f"- Base: `{payload.get('base_id')}`") lines.append(f"- Scanned rows: `{payload.get('scan_limit')}`") lines.append(f"- Interesting payloads: `{len(payload.get('items') or [])}`") lines.append("") lines.append("| File | Bytes | Modified | Format | Named ranges | Sample texts |") lines.append("| --- | --- | --- | --- | --- | --- |") for item in payload.get("items") or []: named_ranges = ", ".join(item.get("named_ranges") or []) sample_texts = ", ".join(item.get("sample_texts") or []) lines.append( f"| `{item.get('file_name')}` | `{item.get('bytes')}` | `{item.get('modified')}` | " f"`{item.get('format')}` | `{named_ranges}` | `{sample_texts}` |" ) return "\n".join(lines) + "\n" def main() -> int: parser = argparse.ArgumentParser(description="Inventory recent 1C template payload candidates from ConfigCAS.") parser.add_argument("--adapter-url", default="http://docker-gpu.cin.su:8011") parser.add_argument("--base-id", default="upo_test") parser.add_argument("--scan-limit", type=int, default=200) parser.add_argument("--max-cells", type=int, default=200) parser.add_argument( "--output-json", default=str(Path("Z:/codex/LLM/reports/1c-template-baselines/upo_test_template_payload_inventory.json")), ) parser.add_argument( "--output-markdown", default=str(Path("Z:/codex/LLM/reports/1c-template-baselines/upo_test_template_payload_inventory.md")), ) args = parser.parse_args() rows = latest_configcas_rows(args.adapter_url, args.base_id, args.scan_limit) items: list[dict[str, Any]] = [] for row in rows: file_name = str(row.get("FileName") or "") byte_count = int(row.get("Bytes") or 0) if not file_name or byte_count <= 0 or byte_count > 20000: continue bundle = read_structure(args.adapter_url, args.base_id, file_name, args.max_cells) summary = summarize_candidate(row, bundle) if is_interesting(summary): items.append(summary) payload = { "schema": "codex_1c_template_payload_inventory.v1", "adapter_url": args.adapter_url, "base_id": args.base_id, "scan_limit": args.scan_limit, "items": items, "counts": {"items": len(items)}, } json_path = Path(args.output_json) md_path = Path(args.output_markdown) json_path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8") md_path.write_text(render_markdown(payload), encoding="utf-8") print( json.dumps( { "status": "ok", "json": str(json_path), "markdown": str(md_path), "items": len(items), }, ensure_ascii=False, indent=2, ) ) return 0 if __name__ == "__main__": raise SystemExit(main())