"""Lossless read-only decoder for 1C Data Composition Schema SQL payloads. The payload stored in ConfigCAS is commonly a compressed stream with a small binary prefix followed by an XML ``SchemaFile`` document. This module does not infer SCD semantics from names: every returned item is backed by an XML node in that document. """ from __future__ import annotations import xml.etree.ElementTree as ET import xml.parsers.expat as expat import html import hashlib import re from typing import Any from .payload import decode_payload_lossless QUERY_PARAMETER_RE = re.compile(r"&([A-Za-z_\u0400-\u04ff][A-Za-z0-9_\u0400-\u04ff]*)") QUERY_SOURCE_RE = re.compile(r"(?:\bИЗ|\bFROM|\bJOIN|\bСОЕДИНЕНИЕ)\s+([A-Za-z_\u0400-\u04ff][A-Za-z0-9_\u0400-\u04ff]*(?:\.[A-Za-z_\u0400-\u04ff][A-Za-z0-9_\u0400-\u04ff]*)+)", re.IGNORECASE) QUERY_SOURCE_BINDING_RE = re.compile(r"(?:\bИЗ|\bFROM|\bJOIN|\bСОЕДИНЕНИЕ)\s+([A-Za-z_\u0400-\u04ff][A-Za-z0-9_\u0400-\u04ff]*(?:\.[A-Za-z_\u0400-\u04ff][A-Za-z0-9_\u0400-\u04ff]*)+)(?:\s+(?:КАК|AS)\s+([A-Za-z_\u0400-\u04ff][A-Za-z0-9_\u0400-\u04ff]*))?", re.IGNORECASE) QUERY_SELECT_RE = re.compile(r"\b(?:ВЫБРАТЬ|SELECT)\b(.*?)(?=\b(?:ИЗ|FROM)\b)", re.IGNORECASE | re.DOTALL) QUERY_ALIAS_RE = re.compile(r"\b(?:КАК|AS)\s+([A-Za-z_\u0400-\u04ff][A-Za-z0-9_\u0400-\u04ff]*)(?=\s*(?:,|\r?\n|$))", re.IGNORECASE) def local_name(tag: str) -> str: return tag.rsplit("}", 1)[-1] if "}" in tag else tag def direct_child(node: ET.Element, name: str) -> ET.Element | None: return next((child for child in node if local_name(child.tag) == name), None) def child_text(node: ET.Element, *names: str) -> str: for name in names: child = direct_child(node, name) if child is not None: value = "".join(child.itertext()).strip() if value: return value return "" def query_without_line_comments(text: str) -> str: """Remove 1C query ``//`` comments without touching quoted string literals.""" result: list[str] = [] index = 0 quoted = False while index < len(text): char = text[index] if char == '"': result.append(char) if quoted and index + 1 < len(text) and text[index + 1] == '"': result.append('"') index += 2 continue quoted = not quoted index += 1 continue if not quoted and char == "/" and index + 1 < len(text) and text[index + 1] == "/": line_end = text.find("\n", index) if line_end < 0: break result.append("\n") index = line_end + 1 continue result.append(char) index += 1 return "".join(result) def node_path(root: ET.Element, target: ET.Element) -> str: """Produce a stable, human-readable evidence path without XML prefixes.""" def visit(node: ET.Element, prefix: str) -> str | None: name = local_name(node.tag) current = f"{prefix}/{name}" if prefix else f"/{name}" if node is target: return current positions: dict[str, int] = {} for child in node: child_name = local_name(child.tag) positions[child_name] = positions.get(child_name, 0) + 1 child_prefix = f"{current}[{positions[child_name]}]" found = visit(child, child_prefix) if found: return found return None return visit(root, "") or "/" def xml_from_scd_payload(data: bytes) -> tuple[ET.Element | None, dict[str, Any]]: decoded = decode_payload_lossless(data) payload = decoded.get("payload") if not isinstance(payload, (bytes, bytearray)): return None, {"status": "undecodable", "code": "SCD_PAYLOAD_EMPTY"} raw = bytes(payload) start = raw.find(b"" end = raw.find(end_marker, start) xml_bytes = raw[start : end + len(end_marker)] if end >= 0 else raw[start:] try: root = ET.fromstring(xml_bytes.decode("utf-8-sig")) except (UnicodeDecodeError, ET.ParseError) as exc: return None, { "status": "undecodable", "code": "SCD_XML_INVALID", "message": str(exc), "compression": decoded.get("compression"), "raw_bytes": decoded.get("raw_bytes"), "payload_bytes": decoded.get("payload_bytes"), } return root, { "status": "ok", "compression": decoded.get("compression"), "raw_bytes": decoded.get("raw_bytes"), "payload_bytes": decoded.get("payload_bytes"), "xml_offset": start, "xml_bytes": len(xml_bytes), "xml_root": local_name(root.tag), } def scd_node_item(root: ET.Element, node: ET.Element, category: str) -> dict[str, Any]: """Return only direct, documented XML values for one SCD item.""" item_name = child_text(node, "name", "dataPath", "field") if not item_name and not list(node): item_name = (node.text or "").strip() item: dict[str, Any] = { "name": item_name, "source": {"kind": "scd_xml", "path": node_path(root, node)}, } expression = child_text(node, "expression") if expression: item["expression"] = expression query = child_text(node, "query") if query: item["query"] = query value_type_node = direct_child(node, "valueType") if value_type_node is None: value_type_node = direct_child(node, "type") value_type = "" if value_type_node is not None: value_type = child_text(value_type_node, "type") or (value_type_node.text or "").strip() if value_type: item["value_type"] = value_type if category == "datasets": item["type"] = node.attrib.get("{http://www.w3.org/2001/XMLSchema-instance}type") or node.attrib.get("type") or "" return item def inspect_scd_payload(data: bytes, *, sections: list[str] | None = None) -> dict[str, Any]: """Decode a DataCompositionSchema XML stream from SQL storage. Unknown or absent XML nodes become empty lists. They are deliberately not synthesized from report code or form attributes. """ requested = sections or ["parameters", "datasets", "fields", "calculated_fields", "resources", "settings", "variants", "total_fields"] root, container = xml_from_scd_payload(data) if root is None: return {"status": "partial", "container": container, "sections": {name: [] for name in requested}} schema = next((node for node in root.iter() if local_name(node.tag) == "dataCompositionSchema"), None) if schema is None: return { "status": "partial", "container": {**container, "code": "SCD_SCHEMA_NODE_NOT_FOUND"}, "sections": {name: [] for name in requested}, } node_names = { "parameters": {"parameter"}, "datasets": {"dataSet"}, "fields": {"field"}, "calculated_fields": {"calculatedField"}, "resources": {"resource"}, "settings": {"settings", "Settings"}, "variants": {"settingsVariant", "variant"}, "total_fields": {"totalField"}, } result: dict[str, list[dict[str, Any]]] = {} skipped_unnamed: dict[str, int] = {} for section in requested: names = node_names.get(section) if not names: result[section] = [] continue raw_items = [scd_node_item(schema, node, section) for node in schema.iter() if local_name(node.tag) in names] result[section] = [item for item in raw_items if item.get("name")] if len(raw_items) != len(result[section]): skipped_unnamed[section] = len(raw_items) - len(result[section]) declared = [str(item.get("name")) for item in result.get("parameters") or [] if item.get("name")] declared_by_normalized = {name.casefold(): name for name in declared} query_references: list[dict[str, Any]] = [] referenced_normalized: set[str] = set() for dataset in result.get("datasets") or []: references: list[str] = [] for found in QUERY_PARAMETER_RE.finditer(query_without_line_comments(str(dataset.get("query") or ""))): name = found.group(1) if name.casefold() not in {value.casefold() for value in references}: references.append(name) referenced_normalized.add(name.casefold()) if references: query_references.append({"dataset": dataset.get("name"), "parameters": references}) analysis = { "kind": "raw_query_parameter_token_scan", "declared_parameters": declared, "query_parameter_references": query_references, "referenced_not_declared_in_schema": sorted( {name for item in query_references for name in item["parameters"] if name.casefold() not in declared_by_normalized}, key=str.casefold, ), "declared_not_referenced_in_dataset_queries": [name for name in declared if name.casefold() not in referenced_normalized], } settings_tags = { "groupings": {"groupItems", "grouping"}, "filters": {"selection", "filter"}, "orders": {"order", "sorting"}, "conditional_appearance": {"appearance", "conditionalAppearance"}, } settings_context: dict[str, Any] = {"status": "not_present", "sections": {}} for context_name, tags in settings_tags.items(): nodes = [node for node in schema.iter() if local_name(node.tag) in tags] if not nodes: continue records: list[dict[str, Any]] = [] for node in nodes: tokens = [] for child in node.iter(): if local_name(child.tag) not in {"field", "dataPath", "left", "right", "group"} or list(child): continue value = (child.text or "").strip() if value and value.casefold() not in {item.casefold() for item in tokens}: tokens.append(value) if tokens: records.append({"path": node_path(schema, node), "tokens": tokens}) if records: settings_context["status"] = "found" settings_context["sections"][context_name] = records analysis["settings_context"] = settings_context query_sources: list[dict[str, Any]] = [] query_output_aliases: list[dict[str, Any]] = [] for dataset in result.get("datasets") or []: query = query_without_line_comments(str(dataset.get("query") or "")) sources = list(dict.fromkeys(match.group(1) for match in QUERY_SOURCE_RE.finditer(query))) if sources: bindings = [] for match in QUERY_SOURCE_BINDING_RE.finditer(query): source, alias = match.group(1), match.group(2) item = {"source": source} if alias: item["alias"] = alias if item not in bindings: bindings.append(item) query_sources.append({"dataset": dataset.get("name"), "sources": sources, "bindings": bindings}) select_match = QUERY_SELECT_RE.search(query) if select_match: aliases = list(dict.fromkeys(match.group(1) for match in QUERY_ALIAS_RE.finditer(select_match.group(1)))) if aliases: query_output_aliases.append({"dataset": dataset.get("name"), "aliases": aliases}) if query_sources: analysis["data_source_references"] = {"kind": "raw_query_source_token_scan", "datasets": query_sources} direct_field_references: list[dict[str, Any]] = [] for dataset in query_sources: query = query_without_line_comments(str(next((item.get("query") for item in result.get("datasets") or [] if item.get("name") == dataset.get("dataset")), ""))) references: list[dict[str, str]] = [] for binding in dataset.get("bindings") or []: alias = str(binding.get("alias") or "") if not alias: continue matcher = re.compile(r"\b" + re.escape(alias) + r"\.([A-Za-z_\u0400-\u04ff][A-Za-z0-9_\u0400-\u04ff]*)(?![A-Za-z0-9_\u0400-\u04ff.])", re.IGNORECASE) for match in matcher.finditer(query): item = {"alias": alias, "field": match.group(1)} if item not in references: references.append(item) if references: direct_field_references.append({"dataset": dataset.get("dataset"), "references": references}) if direct_field_references: analysis["query_direct_field_references"] = {"kind": "direct_alias_field_token_scan", "datasets": direct_field_references} field_names = {str(item.get("name")).casefold(): str(item.get("name")) for item in result.get("fields") or [] if item.get("name")} calculated_field_names = {str(item.get("name")).casefold(): str(item.get("name")) for item in result.get("calculated_fields") or [] if item.get("name")} declared_field_names = {**field_names, **calculated_field_names} total_names = [str(item.get("name")) for item in result.get("total_fields") or [] if item.get("name")] if total_names: analysis["total_field_references"] = { "fields": total_names, "missing_from_declared_fields": [name for name in total_names if name.casefold() not in declared_field_names], "status": "checked" if "fields" in result and "calculated_fields" in result else "field_sections_not_requested", } if query_output_aliases: analysis["query_output_aliases"] = { "kind": "select_clause_alias_scan", "datasets": query_output_aliases, "not_declared_as_scd_fields": sorted( { alias for dataset in query_output_aliases for alias in dataset["aliases"] if alias.casefold() not in declared_field_names }, key=str.casefold, ), } return { "status": "ok", "container": container, "sections": result, "analysis": analysis, "diagnostics": {"skipped_unnamed_xml_nodes": skipped_unnamed} if skipped_unnamed else {}, } def plan_scd_scalar_patch( data: bytes, *, section: str, name: str, property_name: str, value: str, ) -> dict[str, Any]: """Build a byte-preserving patch for one direct scalar SCD XML property. Only query/expression properties are accepted in this first writer layer. The XML element span is collected by Expat from the original byte stream; all bytes outside the scalar content stay unchanged, including the 1C binary prefix/trailer. No database operation is performed here. """ allowed = { "datasets": ({"dataSet"}, {"query"}), "calculated_fields": ({"calculatedField"}, {"expression"}), "resources": ({"resource"}, {"expression"}), } tags_and_properties = allowed.get(section) if not tags_and_properties or property_name not in tags_and_properties[1]: return { "status": "invalid_argument", "code": "SCD_PATCH_PROPERTY_UNSUPPORTED", "message": "Only datasets.query, calculated_fields.expression, and resources.expression are writable.", } root, container = xml_from_scd_payload(data) if root is None: return {"status": "undecodable", "container": container} decoded = decode_payload_lossless(data) payload = bytes(decoded["payload"]) xml_start = payload.find(b"" xml_end = payload.find(xml_end_marker, xml_start) if xml_start < 0 or xml_end < 0: return {"status": "undecodable", "container": container} xml_end += len(xml_end_marker) xml = payload[xml_start:xml_end] target_tags = tags_and_properties[0] stack: list[dict[str, Any]] = [] records: list[dict[str, Any]] = [] def start_element(tag: str, _attrs: dict[str, str]) -> None: local = local_name(tag) position = parser.CurrentByteIndex end = xml.find(b">", position) frame: dict[str, Any] = {"tag": local, "depth": len(stack) + 1, "content_start": end + 1} if local in target_tags: frame["record"] = {"tag": local, "depth": len(stack) + 1, "properties": {}} if stack: parent_record = next((item.get("record") for item in reversed(stack) if item.get("record")), None) if parent_record and len(stack) + 1 == parent_record["depth"] + 1 and local in {"name", "dataPath", property_name}: frame["property_record"] = parent_record stack.append(frame) def end_element(_tag: str) -> None: frame = stack.pop() end = parser.CurrentByteIndex property_record = frame.get("property_record") if property_record is not None: raw_text = xml[int(frame["content_start"]):end] if b"<" not in raw_text: property_record["properties"][frame["tag"]] = { "start": int(frame["content_start"]), "end": end, "text": html.unescape(raw_text.decode("utf-8")), } record = frame.get("record") if record is not None: identity = record["properties"].get("name") or record["properties"].get("dataPath") record["name"] = identity.get("text") if identity else "" records.append(record) parser = expat.ParserCreate() parser.StartElementHandler = start_element parser.EndElementHandler = end_element try: parser.Parse(xml, True) except expat.ExpatError as exc: return {"status": "undecodable", "container": container, "code": "SCD_XML_INVALID", "message": str(exc)} matches = [record for record in records if str(record.get("name") or "") == name] if not matches: return {"status": "not_found", "code": "SCD_PATCH_TARGET_NOT_FOUND", "container": container} if len(matches) > 1: return {"status": "ambiguous", "code": "SCD_PATCH_TARGET_AMBIGUOUS", "container": container, "matches": len(matches)} property_record = (matches[0].get("properties") or {}).get(property_name) if not property_record: return {"status": "not_found", "code": "SCD_PATCH_PROPERTY_NOT_FOUND", "container": container} old = str(property_record["text"]) if old == value: return {"status": "unchanged", "container": container, "old": old, "new": value} escaped = html.escape(value, quote=False).encode("utf-8") patched_xml = xml[: property_record["start"]] + escaped + xml[property_record["end"] :] patched_payload = payload[:xml_start] + patched_xml + payload[xml_end:] from .payload import encode_payload_lossless patched_data = encode_payload_lossless(decoded, payload=patched_payload) return { "status": "planned", "container": container, "old": old, "new": value, "payload": patched_data, "expected_sha1": hashlib.sha1(data).hexdigest(), "result_sha1": hashlib.sha1(patched_data).hexdigest(), "changed_bytes": len(patched_data) - len(data), } def compare_scd_semantics(active: dict[str, Any], saved: dict[str, Any]) -> dict[str, Any]: """Compare decoded SCD sections by semantic content, never by storage id.""" section_names = sorted(set((active.get("sections") or {}).keys()) | set((saved.get("sections") or {}).keys())) sections: dict[str, dict[str, Any]] = {} counts = {"added": 0, "removed": 0, "changed": 0, "unchanged": 0} for section in section_names: def index(items: Any) -> dict[str, Any]: result: dict[str, Any] = {} for ordinal, item in enumerate(items or []): if not isinstance(item, dict): continue key = str(item.get("name") or f"#{ordinal}") result[key] = {key: value for key, value in item.items() if key != "source"} return result active_items, saved_items = index((active.get("sections") or {}).get(section)), index((saved.get("sections") or {}).get(section)) added = sorted(set(saved_items) - set(active_items), key=str.casefold) removed = sorted(set(active_items) - set(saved_items), key=str.casefold) changed = sorted([name for name in set(active_items) & set(saved_items) if active_items[name] != saved_items[name]], key=str.casefold) unchanged = len(set(active_items) & set(saved_items)) - len(changed) sections[section] = {"added": added, "removed": removed, "changed": changed, "unchanged": unchanged} counts["added"] += len(added); counts["removed"] += len(removed); counts["changed"] += len(changed); counts["unchanged"] += unchanged return {"status": "unchanged" if not any(counts[key] for key in ("added", "removed", "changed")) else "changed", "sections": sections, "counts": counts}