Initial SQL-only 1C adapter baseline

This commit is contained in:
2026-07-22 03:03:47 +03:00
commit e2503b77e7
545 changed files with 184711 additions and 0 deletions
+191
View File
@@ -0,0 +1,191 @@
#!/usr/bin/env python3
"""Describe extension manifest payloads by observable structure.
This script intentionally reports structural facts first. Semantic labels are
only added when the evidence is direct, for example a payload contains BSL text
or embedded HTML help.
"""
from __future__ import annotations
import argparse
import base64
import json
import re
from collections import Counter, defaultdict
from pathlib import Path
from typing import Any
from inspect_1c_sql_files import Lexer, Parser, collect_strings, tree_shape, try_decode, try_decompress
BSL_MARKERS = ("&На", "Процедура ", "Функция ", "#Область", "#КонецОбласти")
HTML_MARKERS = ("<!DOCTYPE HTML", "<html", "<HTML", "<body", "<BODY")
def scalar(node: Any) -> str:
if isinstance(node, dict) and node.get("type") in {"atom", "string"}:
return str(node.get("value") or "")
return ""
def suffix_of(object_id: str) -> str:
parts = object_id.split(".", 1)
return "" if len(parts) == 1 else "." + parts[1]
def parse_payload(path: Path) -> dict[str, Any]:
raw = path.read_bytes()
payload, compression = try_decompress(raw)
text, encoding = try_decode(payload)
result: dict[str, Any] = {
"bytes": len(raw),
"payload_bytes": len(payload),
"compression": compression,
"encoding": encoding,
"parse_status": "not_text",
"root_kind": "",
"root_len": None,
"root_marker": "",
"strings_sample": [],
"bsl_marker_count": 0,
"html_marker_count": 0,
"base64_atom_count": 0,
"embedded_base64_html_count": 0,
"semantic_evidence": [],
}
if text is None:
return result
clean = text.replace("\x00", "").replace("\ufeff", "").lstrip("ï»¿п»ї")
result["bsl_marker_count"] = sum(clean.count(marker) for marker in BSL_MARKERS)
result["html_marker_count"] = sum(clean.count(marker) for marker in HTML_MARKERS)
try:
parsed = Parser(Lexer(clean[:2_000_000]).tokens()).parse()
except Exception as exc:
result["parse_status"] = "parse_error"
result["parse_error"] = str(exc)
return result
result["parse_status"] = "parsed"
result["shape"] = tree_shape(parsed, max_depth=3)
strings = collect_strings(parsed, limit=80)
result["strings_sample"] = strings[:40]
if isinstance(parsed, dict):
result["root_kind"] = parsed.get("type") or ""
items = parsed.get("items") or []
result["root_len"] = len(items)
if items:
result["root_marker"] = scalar(items[0])
atoms = []
def walk(node: Any) -> None:
if isinstance(node, dict) and node.get("type") == "atom":
atoms.append(str(node.get("value") or ""))
if isinstance(node, dict):
for child in node.get("items") or []:
walk(child)
walk(parsed)
b64_atoms = [value for value in atoms if re.fullmatch(r"[A-Za-z0-9+/]{40,}={0,2}", value)]
result["base64_atom_count"] = len(b64_atoms)
html_count = 0
for value in b64_atoms[:200]:
try:
decoded = base64.b64decode(value, validate=False)
except Exception:
continue
if any(marker.encode("utf-8") in decoded or marker.encode("cp1251", errors="ignore") in decoded for marker in HTML_MARKERS):
html_count += 1
result["embedded_base64_html_count"] = html_count
evidence = []
if result["bsl_marker_count"]:
evidence.append("contains_bsl_text")
if result["html_marker_count"] or html_count:
evidence.append("contains_html")
if result["base64_atom_count"]:
evidence.append("contains_base64_atoms")
result["semantic_evidence"] = evidence
return result
def main() -> int:
parser = argparse.ArgumentParser(description="Classify manifest payloads by observed structure.")
parser.add_argument("--manifest-dir", type=Path, required=True)
parser.add_argument("--cas-dir", type=Path, required=True)
parser.add_argument("--xml-index", type=Path)
parser.add_argument("--output", type=Path, required=True)
parser.add_argument("--limit", type=int, default=0)
args = parser.parse_args()
xml_map = {}
if args.xml_index and args.xml_index.is_file():
xml = json.loads(args.xml_index.read_text(encoding="utf-8"))
xml_map = xml.get("guid_map") or {}
entries = []
for manifest_path in sorted(args.manifest_dir.glob("*.json")):
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
extension_file = (manifest.get("extension_zipped_info") or {}).get("file_name") or manifest_path.name
for entry in manifest.get("entries") or []:
cas_path = Path(entry.get("cas_path") or args.cas_dir / entry["cas_key"])
if not cas_path.is_file():
continue
object_id = entry["object_id"]
base_guid = object_id.split(".", 1)[0].lower()
payload = parse_payload(cas_path)
xml_item = xml_map.get(base_guid) or {}
top_objects = xml_item.get("top_objects") or []
entries.append(
{
"extension_file": extension_file,
"manifest_path": str(manifest_path),
"object_id": object_id,
"base_guid": base_guid,
"suffix": suffix_of(object_id),
"cas_key": entry["cas_key"],
"xml_top_objects": top_objects[:5],
"payload": payload,
}
)
if args.limit and len(entries) >= args.limit:
break
if args.limit and len(entries) >= args.limit:
break
suffix_counts = Counter(item["suffix"] for item in entries)
suffix_root_counts: dict[str, Counter[str]] = defaultdict(Counter)
suffix_evidence_counts: dict[str, Counter[str]] = defaultdict(Counter)
for item in entries:
suffix = item["suffix"]
payload = item["payload"]
root_signature = f"{payload.get('root_kind')}:{payload.get('root_marker')}:{payload.get('root_len')}"
suffix_root_counts[suffix][root_signature] += 1
for evidence in payload.get("semantic_evidence") or ["<none>"]:
suffix_evidence_counts[suffix][evidence] += 1
report = {
"schema": "onec_manifest_payload_structure.v1",
"manifest_dir": str(args.manifest_dir),
"cas_dir": str(args.cas_dir),
"entry_count": len(entries),
"suffix_counts": dict(sorted(suffix_counts.items())),
"suffix_root_counts": {key: dict(value.most_common()) for key, value in sorted(suffix_root_counts.items())},
"suffix_evidence_counts": {key: dict(value.most_common()) for key, value in sorted(suffix_evidence_counts.items())},
"entries": entries,
}
args.output.parent.mkdir(parents=True, exist_ok=True)
args.output.write_text(json.dumps(report, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
print(
json.dumps(
{
"output": str(args.output),
"entries": len(entries),
"suffixes": len(suffix_counts),
},
ensure_ascii=False,
)
)
return 0
if __name__ == "__main__":
raise SystemExit(main())