#!/usr/bin/env python3 """Inspect exported 1C SQL Config/Params files without semantic guessing. This script operates on files exported from Config/ConfigSave/ConfigCAS/ ConfigCASSave/Params tables. It only performs mechanical steps: 1. try known compression envelopes; 2. try common text encodings; 3. parse the brace-based 1C serialized value syntax into a generic tree; 4. report structural shape, strings, GUIDs, and selected tree paths. It deliberately does not map SQLKind codes to 1C object types. """ from __future__ import annotations import argparse import gzip import json import re import zlib from dataclasses import dataclass from pathlib import Path from typing import Any GUID_RE = re.compile( r"[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}" ) def try_decompress(data: bytes) -> tuple[bytes, str]: attempts = ( ("raw_deflate", lambda value: zlib.decompress(value, -15)), ("zlib", zlib.decompress), ("gzip", gzip.decompress), ) for name, func in attempts: try: return func(data), name except Exception: pass return data, "none" def try_decode(data: bytes) -> tuple[str | None, str | None]: candidates = ("utf-8-sig", "utf-16-le", "utf-16-be", "cp1251") best: tuple[str | None, str | None, int] = (None, None, -1) for encoding in candidates: try: text = data.decode(encoding) except UnicodeDecodeError: continue printable = sum(1 for char in text[:20000] if char.isprintable() or char in "\r\n\t") nul_count = text[:20000].count("\x00") score = printable - nul_count * 10 if score > best[2]: best = (text, encoding, score) return best[0], best[1] @dataclass class Token: kind: str value: str pos: int class Lexer: def __init__(self, text: str) -> None: self.text = text self.pos = 0 def tokens(self) -> list[Token]: result: list[Token] = [] while self.pos < len(self.text): char = self.text[self.pos] if char.isspace(): self.pos += 1 continue if char in "{},:": result.append(Token(char, char, self.pos)) self.pos += 1 continue if char == '"': result.append(self._string()) continue result.append(self._atom()) result.append(Token("EOF", "", self.pos)) return result def _string(self) -> Token: start = self.pos self.pos += 1 chars: list[str] = [] while self.pos < len(self.text): char = self.text[self.pos] self.pos += 1 if char == '"': if self.pos < len(self.text) and self.text[self.pos] == '"': chars.append('"') self.pos += 1 continue break chars.append(char) return Token("string", "".join(chars), start) def _atom(self) -> Token: start = self.pos while self.pos < len(self.text): char = self.text[self.pos] if char.isspace() or char in "{},:": break self.pos += 1 return Token("atom", self.text[start : self.pos], start) class Parser: def __init__(self, tokens: list[Token]) -> None: self.tokens = tokens self.index = 0 def parse(self) -> Any: values = [] while not self._peek("EOF"): if self._peek(","): self.index += 1 continue values.append(self._value()) if len(values) == 1: return values[0] return {"type": "sequence", "items": values} def _value(self) -> Any: if self._peek("{"): return self._list() token = self._next() if token.kind == "string": return {"type": "string", "value": token.value} if token.kind == "atom": return {"type": "atom", "value": token.value} return {"type": "token", "kind": token.kind, "value": token.value} def _list(self) -> Any: start = self._next() items = [] while not self._peek("EOF") and not self._peek("}"): if self._peek(","): self.index += 1 continue items.append(self._value()) if self._peek("}"): self.index += 1 return {"type": "list", "pos": start.pos, "items": items} def _peek(self, kind: str) -> bool: return self.tokens[self.index].kind == kind def _next(self) -> Token: token = self.tokens[self.index] self.index += 1 return token def tree_shape(value: Any, depth: int = 0, max_depth: int = 4) -> Any: if depth >= max_depth: return "..." if isinstance(value, dict) and value.get("type") == "list": items = value.get("items") or [] return { "type": "list", "len": len(items), "items": [tree_shape(item, depth + 1, max_depth) for item in items[:12]], } if isinstance(value, dict) and value.get("type") in {"string", "atom"}: raw = str(value.get("value") or "") return {"type": value["type"], "value": raw[:120], "len": len(raw)} if isinstance(value, dict) and value.get("type") == "sequence": items = value.get("items") or [] return { "type": "sequence", "len": len(items), "items": [tree_shape(item, depth + 1, max_depth) for item in items[:12]], } return str(value)[:120] def collect_strings(value: Any, limit: int = 200) -> list[str]: result: list[str] = [] def walk(node: Any) -> None: if len(result) >= limit: return if isinstance(node, dict) and node.get("type") == "string": text = str(node.get("value") or "") if text: result.append(text) return if isinstance(node, dict): for child in node.get("items") or []: walk(child) walk(value) return result def inspect_file(path: Path, *, parse_limit_chars: int) -> dict[str, Any]: original = path.read_bytes() payload, compression = try_decompress(original) text, encoding = try_decode(payload) result: dict[str, Any] = { "file_name": path.name, "bytes": len(original), "payload_bytes": len(payload), "compression": compression, "encoding": encoding, "guids": [], "strings": [], "parse_status": "not_text", } if text is None: return result clean = text.replace("\x00", "").replace("\ufeff", "") result["text_preview"] = clean[:500] result["guids"] = sorted(set(match.lower() for match in GUID_RE.findall(clean)))[:200] if "{" not in clean: result["parse_status"] = "text_no_braces" return result parse_text = clean[:parse_limit_chars] try: parsed = Parser(Lexer(parse_text).tokens()).parse() except Exception as exc: result["parse_status"] = "parse_error" result["parse_error"] = str(exc) return result result["parse_status"] = "parsed_prefix" if len(clean) > parse_limit_chars else "parsed" result["shape"] = tree_shape(parsed) result["strings"] = collect_strings(parsed) return result def main() -> int: parser = argparse.ArgumentParser(description="Inspect exported 1C SQL files mechanically.") parser.add_argument("input", type=Path, help="Directory with exported SQL files") parser.add_argument("--output", type=Path, required=True, help="JSON report path") parser.add_argument("--parse-limit-chars", type=int, default=2_000_000) parser.add_argument("--limit", type=int, default=0) args = parser.parse_args() files = sorted(path for path in args.input.iterdir() if path.is_file()) if args.limit > 0: files = files[: args.limit] report = { "schema": "onec_sql_file_inspection.v1", "input": str(args.input), "file_count": len(files), "files": [inspect_file(path, parse_limit_chars=args.parse_limit_chars) for path in files], } args.output.parent.mkdir(parents=True, exist_ok=True) args.output.write_text(json.dumps(report, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") print(json.dumps({"output": str(args.output), "files": len(files)}, ensure_ascii=False)) return 0 if __name__ == "__main__": raise SystemExit(main())