#!/usr/bin/env python3 """Compare one extension manifest object parts with an XML export tree. The script is intentionally evidence-first. It reports mechanical facts: manifest object_id parts, ConfigCAS keys, decoded payload structure, embedded base64 blobs, and exact/contains matches against files from an XML export. It does not assign semantic names to suffixes. """ from __future__ import annotations import argparse import base64 import hashlib import json import re from pathlib import Path from typing import Any from inspect_1c_sql_files import GUID_RE, Lexer, Parser, collect_strings, tree_shape, try_decode, try_decompress BASE64_RE = re.compile(r"[A-Za-z0-9+/]{40,}={0,2}") BSL_MARKERS = ("&На", "Процедура ", "Функция ", "#Область", "#КонецОбласти") HTML_MARKERS = (" str: if isinstance(node, dict) and node.get("type") in {"atom", "string"}: return str(node.get("value") or "") return "" def suffix_of(object_id: str) -> str: parts = object_id.split(".", 1) return "" if len(parts) == 1 else "." + parts[1] def sha1_hex(data: bytes) -> str: return hashlib.sha1(data).hexdigest() def decode_text(data: bytes) -> tuple[str | None, str | None]: if data.startswith(b"\xef\xbb\xbf"): try: return data.decode("utf-8-sig"), "utf-8-sig" except UnicodeDecodeError: pass return try_decode(data) def decode_payload_text(data: bytes) -> tuple[str | None, str | None, int]: if data.startswith(b"\xef\xbb\xbf"): text, encoding = decode_text(data) return text, encoding, 0 marker = data.find(b"\xef\xbb\xbf") if marker >= 0: try: return data[marker:].decode("utf-8-sig"), "utf-8-sig", marker except UnicodeDecodeError: pass text, encoding = try_decode(data) return text, encoding, 0 def payload_markers(data: bytes) -> list[str]: markers = [] if data.startswith(b"MOXCEL"): markers.append("MOXCEL") if data.startswith(b"\xef\xbb\xbf") or b"\xef\xbb\xbf" in data[:256]: markers.append("utf8_bom") if STREAM_HEADER_RE.search(data): markers.append("stream_headers") return markers def load_xml_files(paths: list[Path]) -> list[dict[str, Any]]: files: list[dict[str, Any]] = [] seen: set[Path] = set() for root in paths: if not root.exists(): continue candidates = [root] if root.is_file() else [item for item in root.rglob("*") if item.is_file()] for path in candidates: resolved = path.resolve() if resolved in seen: continue seen.add(resolved) data = path.read_bytes() text, encoding = decode_text(data) files.append( { "path": str(path), "name": path.name, "relative_hint": str(path), "bytes": len(data), "sha1": sha1_hex(data), "text": text, "encoding": encoding, } ) return files def collect_atoms(value: Any) -> list[str]: atoms: list[str] = [] def walk(node: Any) -> None: if isinstance(node, dict) and node.get("type") == "atom": atoms.append(str(node.get("value") or "")) if isinstance(node, dict): for child in node.get("items") or []: walk(child) walk(value) return atoms def collect_base64_blocks(value: Any) -> list[str]: blocks: list[str] = [] def walk(node: Any) -> None: if isinstance(node, dict) and node.get("type") == "list": items = node.get("items") or [] if items and scalar(items[0]) == "#base64": chunks = [scalar(item) for item in items[1:] if BASE64_RE.fullmatch(scalar(item))] if chunks: blocks.append("".join(chunks)) for child in items: walk(child) walk(value) return blocks def decode_base64_atoms(atoms: list[str]) -> list[dict[str, Any]]: decoded: list[dict[str, Any]] = [] for value in atoms: if not BASE64_RE.fullmatch(value): continue try: data = base64.b64decode(value, validate=True) except Exception: continue if not data: continue text, encoding = decode_text(data) text_preview = "" if text: text_preview = text.replace("\x00", "")[:300] decoded.append( { "atom_length": len(value), "bytes": len(data), "sha1": sha1_hex(data), "encoding": encoding, "text_preview": text_preview, "has_bsl_marker": bool(text and any(marker in text for marker in BSL_MARKERS)), "has_html_marker": bool(text and any(marker in text for marker in HTML_MARKERS)), "bytes_base64": value[:120], "data": data, "text": text, } ) return decoded def decode_base64_blocks(blocks: list[str]) -> list[dict[str, Any]]: decoded: list[dict[str, Any]] = [] for value in blocks: try: data = base64.b64decode(value, validate=True) except Exception: continue text, encoding = decode_text(data) decoded.append( { "block_length": len(value), "bytes": len(data), "sha1": sha1_hex(data), "encoding": encoding, "text_preview": (text or "").replace("\x00", "")[:500], "has_bsl_marker": bool(text and any(marker in text for marker in BSL_MARKERS)), "has_html_marker": bool(text and any(marker in text for marker in HTML_MARKERS)), "data": data, "text": text, } ) return decoded def match_blob(blob: bytes, text: str | None, xml_files: list[dict[str, Any]]) -> dict[str, Any]: exact = [] normalized_equal = [] contains = [] blob_sha1 = sha1_hex(blob) normalized_text = text.replace("\r\n", "\n").strip() if text else None for item in xml_files: file_text = item.get("text") data = Path(item["path"]).read_bytes() if item["sha1"] == blob_sha1: exact.append({"path": item["path"], "match": "sha1"}) elif normalized_text and file_text and normalized_text == file_text.replace("\r\n", "\n").strip(): normalized_equal.append({"path": item["path"], "match": "normalized_text_equal"}) elif len(blob) >= 24 and blob in data: contains.append({"path": item["path"], "match": "bytes_contains"}) elif text and file_text and len(text.strip()) >= 24 and text.strip() in file_text: contains.append({"path": item["path"], "match": "text_contains"}) elif text and file_text and len(file_text.strip()) >= 24 and file_text.strip() in text: contains.append({"path": item["path"], "match": "payload_contains_file_text"}) return {"exact": exact, "normalized_equal": normalized_equal, "contains": contains} def extract_stream_blocks(payload: bytes) -> list[dict[str, Any]]: blocks: list[dict[str, Any]] = [] for match in STREAM_HEADER_RE.finditer(payload): declared_1 = int(match.group(1), 16) declared_2 = int(match.group(2), 16) start = match.end() size = declared_2 if size <= 0 or start + size > len(payload): continue data = payload[start : start + size] text, encoding = decode_text(data) blocks.append( { "header_offset": match.start(), "data_offset": start, "declared_1": declared_1, "declared_2": declared_2, "bytes": len(data), "sha1": sha1_hex(data), "encoding": encoding, "text_preview": (text or "").replace("\x00", "")[:500], "has_bsl_marker": bool(text and any(marker in text for marker in BSL_MARKERS)), "has_html_marker": bool(text and any(marker in text for marker in HTML_MARKERS)), "data": data, "text": text, } ) return blocks def match_strings(strings: list[str], xml_files: list[dict[str, Any]], *, limit: int = 80) -> list[dict[str, Any]]: hits: list[dict[str, Any]] = [] for string in strings: if len(string.strip()) < 4: continue paths = [] for item in xml_files: text = item.get("text") or "" if string in text: paths.append(item["path"]) if len(paths) >= 8: break if paths: hits.append({"string": string[:200], "paths": paths}) if len(hits) >= limit: break return hits def match_guids(text: str, xml_files: list[dict[str, Any]], *, limit: int = 200) -> list[dict[str, Any]]: hits: list[dict[str, Any]] = [] for guid in sorted(set(match.lower() for match in GUID_RE.findall(text))): paths = [] for item in xml_files: file_text = (item.get("text") or "").lower() if guid in file_text: paths.append(item["path"]) if len(paths) >= 8: break if paths: hits.append({"guid": guid, "paths": paths}) if len(hits) >= limit: break return hits def parse_cas_payload(path: Path) -> dict[str, Any]: raw = path.read_bytes() payload, compression = try_decompress(raw) markers = payload_markers(payload) if "stream_headers" in markers and not payload.startswith(b"\xef\xbb\xbf"): text, encoding, text_offset = None, None, 0 else: text, encoding, text_offset = decode_payload_text(payload) report: dict[str, Any] = { "cas_file": path.name, "bytes": len(raw), "payload_bytes": len(payload), "payload_sha1": sha1_hex(payload), "compression": compression, "encoding": encoding, "text_offset": text_offset, "payload_markers": markers, "parse_status": "not_text", "text_preview": "", "strings": [], "base64_blobs": [], "base64_blocks": [], "stream_blocks": [], } report["_raw_payload"] = payload report["_stream_blocks"] = extract_stream_blocks(payload) if text is None: return report clean = text.replace("\x00", "").replace("\ufeff", "").lstrip("ï»¿п»ї") report["text_preview"] = clean[:500] report["has_bsl_marker"] = any(marker in clean for marker in BSL_MARKERS) report["has_html_marker"] = any(marker in clean for marker in HTML_MARKERS) if "{" not in clean: report["parse_status"] = "text_no_braces" return report try: parsed = Parser(Lexer(clean[:2_000_000]).tokens()).parse() except Exception as exc: report["parse_status"] = "parse_error" report["parse_error"] = str(exc) return report report["parse_status"] = "parsed" report["shape"] = tree_shape(parsed, max_depth=4) strings = collect_strings(parsed, limit=200) atoms = collect_atoms(parsed) blocks = collect_base64_blocks(parsed) report["strings"] = strings if isinstance(parsed, dict): items = parsed.get("items") or [] report["root_type"] = parsed.get("type") report["root_len"] = len(items) report["root_marker"] = scalar(items[0]) if items else "" report["_raw_payload"] = payload report["_clean_text"] = clean report["_base64_decoded"] = decode_base64_atoms(atoms) report["_base64_blocks_decoded"] = decode_base64_blocks(blocks) return report def load_manifest_entries(manifest_path: Path, object_guid: str) -> list[dict[str, Any]]: manifest = json.loads(manifest_path.read_text(encoding="utf-8")) target = object_guid.lower() entries = [] for entry in manifest.get("entries") or []: object_id = str(entry.get("object_id") or "").lower() if object_id == target or object_id.startswith(target + "."): entries.append(entry) entries.sort(key=lambda item: (suffix_of(str(item["object_id"])), str(item["object_id"]))) return entries def public_payload_report(payload_report: dict[str, Any], xml_files: list[dict[str, Any]]) -> dict[str, Any]: raw_payload = payload_report.pop("_raw_payload", b"") clean_text = payload_report.pop("_clean_text", "") decoded = payload_report.pop("_base64_decoded", []) decoded_blocks = payload_report.pop("_base64_blocks_decoded", []) stream_blocks = payload_report.pop("_stream_blocks", []) payload_report["payload_matches"] = match_blob(raw_payload, clean_text, xml_files) payload_report["string_matches"] = match_strings(payload_report.get("strings") or [], xml_files) payload_report["guid_matches"] = match_guids(clean_text, xml_files) public_decoded = [] for blob in decoded: data = blob.pop("data") text = blob.pop("text") blob["matches"] = match_blob(data, text, xml_files) public_decoded.append(blob) payload_report["base64_blobs"] = public_decoded public_blocks = [] for blob in decoded_blocks: data = blob.pop("data") text = blob.pop("text") blob["matches"] = match_blob(data, text, xml_files) public_blocks.append(blob) payload_report["base64_blocks"] = public_blocks public_streams = [] for block in stream_blocks: data = block.pop("data") text = block.pop("text") block["matches"] = match_blob(data, text, xml_files) public_streams.append(block) payload_report["stream_blocks"] = public_streams payload_report["strings"] = (payload_report.get("strings") or [])[:80] return payload_report def main() -> int: parser = argparse.ArgumentParser(description="Compare one manifest object's CAS parts with XML files.") parser.add_argument("--manifest", type=Path, required=True) parser.add_argument("--cas-dir", type=Path, required=True) parser.add_argument("--object-guid", required=True) parser.add_argument("--xml-path", type=Path, action="append", default=[]) parser.add_argument("--output", type=Path, required=True) args = parser.parse_args() xml_files = load_xml_files(args.xml_path) entries = load_manifest_entries(args.manifest, args.object_guid) parts = [] for entry in entries: cas_key = entry["cas_key"] cas_path = Path(entry.get("cas_path") or args.cas_dir / cas_key) if not cas_path.is_file(): cas_path = args.cas_dir / cas_key payload = parse_cas_payload(cas_path) if cas_path.is_file() else {"parse_status": "missing_cas"} parts.append( { "object_id": entry["object_id"], "suffix": suffix_of(entry["object_id"]), "cas_key": cas_key, "cas_path": str(cas_path), "payload": public_payload_report(payload, xml_files) if cas_path.is_file() else payload, } ) report = { "schema": "onec_manifest_object_part_compare.v1", "object_guid": args.object_guid.lower(), "manifest": str(args.manifest), "cas_dir": str(args.cas_dir), "xml_paths": [str(path) for path in args.xml_path], "xml_file_count": len(xml_files), "xml_files": [ { "path": item["path"], "bytes": item["bytes"], "sha1": item["sha1"], "encoding": item["encoding"], } for item in xml_files ], "part_count": len(parts), "parts": parts, } args.output.parent.mkdir(parents=True, exist_ok=True) args.output.write_text(json.dumps(report, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") print(json.dumps({"output": str(args.output), "parts": len(parts), "xml_files": len(xml_files)}, ensure_ascii=False)) return 0 if __name__ == "__main__": raise SystemExit(main())