Initial SQL-only 1C adapter baseline

This commit is contained in:
2026-07-22 03:03:47 +03:00
commit e2503b77e7
545 changed files with 184711 additions and 0 deletions
+271
View File
@@ -0,0 +1,271 @@
"""Lightweight structural checks for 1C BSL text."""
from __future__ import annotations
import re
import hashlib
from typing import Any
WORD = r"А-Яа-яA-Za-z0-9_"
ROUTINE_START_RE = re.compile(r"(?im)^\s*(?:Асинх\s+)?(Процедура|Функция)\s+([A-Za-zА-Яа-яЁё_][\wА-Яа-яЁё]*)\s*\(")
ROUTINE_END_RE = re.compile(r"(?im)^\s*(КонецПроцедуры|КонецФункции)\b")
ROUTINE_RE = re.compile(rf"(?im)^\s*(?:Асинх\s+)?(Процедура|Функция)\s+([А-Яа-яA-Za-z_][{WORD}]*)\s*\(")
END_RE = {
"процедура": re.compile(rf"(?<![{WORD}])КонецПроцедуры(?![{WORD}])", re.IGNORECASE),
"функция": re.compile(rf"(?<![{WORD}])КонецФункции(?![{WORD}])", re.IGNORECASE),
}
REGION_START_RE = re.compile(r"(?im)^\s*#Область\b")
REGION_END_RE = re.compile(r"(?im)^\s*#КонецОбласти\b")
PREPROC_IF_RE = re.compile(r"(?im)^\s*#Если\b")
PREPROC_ENDIF_RE = re.compile(r"(?im)^\s*#КонецЕсли\b")
def normalize_name(value: str | None) -> str:
return re.sub(r"[\s._-]+", "", str(value or "")).casefold()
def line_starts(text: str) -> list[int]:
starts = [0]
for match in re.finditer(r"\n", text):
starts.append(match.end())
return starts
def offset_to_line(starts: list[int], offset: int) -> int:
line = 1
for index, start in enumerate(starts, start=1):
if start > offset:
break
line = index
return line
def routine_blocks(text: str) -> list[dict[str, Any]]:
blocks = []
lines = text.splitlines(keepends=True)
line_offsets: list[int] = []
offset = 0
for line in lines:
line_offsets.append(offset)
offset += len(line)
for line_index, line in enumerate(lines):
code = strip_line_comment(line)
match = ROUTINE_RE.match(code)
if not match:
continue
kind = match.group(1)
name = match.group(2)
declaration_start = line_offsets[line_index] + match.start()
end = len(text)
line_end = len(lines) or 1
end_re = END_RE[kind.casefold()]
for end_line_index in range(line_index + 1, len(lines)):
end_code = strip_line_comment(lines[end_line_index])
end_match = end_re.search(end_code)
if end_match:
end = line_offsets[end_line_index] + end_match.end()
line_end = end_line_index + 1
break
blocks.append(
{
"kind": kind,
"name": name,
"normalized_name": normalize_name(name),
"start": declaration_start,
"declaration_start": declaration_start,
"end": end,
"line_start": line_index + 1,
"line_end": line_end,
}
)
return blocks
def directive_start(text: str, declaration_start: int) -> int:
prefix = text[:declaration_start]
lines = prefix.splitlines(keepends=True)
start_offset = len(prefix)
index = len(lines) - 1
while index >= 0:
line = lines[index]
stripped = line.strip()
if stripped.startswith("&"):
start_offset -= len(line)
index -= 1
continue
if stripped == "":
candidate = index - 1
while candidate >= 0 and lines[candidate].strip() == "":
candidate -= 1
if candidate >= 0 and lines[candidate].strip().startswith("&"):
start_offset -= len(line)
index -= 1
continue
break
return start_offset
def one_routine_from_text(routine_text: str) -> dict[str, Any]:
blocks = routine_blocks(routine_text)
if len(blocks) != 1:
raise ValueError(f"routine_text must contain exactly one procedure/function, found {len(blocks)}")
block = blocks[0]
if block["end"] < len(routine_text.rstrip()):
suffix = routine_text[block["end"] :].strip()
if suffix:
raise ValueError("routine_text must not contain extra code after the routine end")
return block
def text_sha1(value: str) -> str:
return hashlib.sha1(value.encode("utf-8")).hexdigest()
def dominant_eol(text: str) -> str:
crlf = text.count("\r\n")
without_crlf = text.replace("\r\n", "")
lf = without_crlf.count("\n")
cr = without_crlf.count("\r")
if crlf >= lf and crlf >= cr and crlf > 0:
return "\r\n"
if cr > lf and cr > 0:
return "\r"
return "\n"
def normalize_eol(text: str, eol: str) -> str:
normalized = text.replace("\r\n", "\n").replace("\r", "\n")
return normalized.replace("\n", eol)
def replace_routine_text(
text: str,
routine_text: str,
*,
operation: str = "replace",
name: str | None = None,
expected_old_sha1: str | None = None,
expected_old_contains: str | None = None,
) -> tuple[str, dict[str, Any]]:
if operation not in {"replace", "append", "upsert"}:
raise ValueError("routine operation must be replace, append, or upsert")
new_block = one_routine_from_text(routine_text)
wanted = normalize_name(name or new_block["name"])
blocks = routine_blocks(text)
matches = [block for block in blocks if block["normalized_name"] == wanted]
if len(matches) > 1:
raise ValueError(f"target module has duplicate routine: {name or new_block['name']}")
exists = bool(matches)
if operation == "append" and exists:
raise ValueError(f"routine already exists: {new_block['name']}")
if operation == "replace" and not exists:
raise ValueError(f"routine does not exist: {name or new_block['name']}")
eol = dominant_eol(text)
replacement = normalize_eol(routine_text.strip(), eol)
if exists:
old = matches[0]
start = directive_start(text, int(old["declaration_start"]))
end = int(old["end"])
old_text = text[start:end]
old_sha1 = text_sha1(old_text)
if expected_old_sha1 and expected_old_sha1.lower() != old_sha1:
raise ValueError("routine expected_old_sha1 does not match current routine text")
if expected_old_contains and expected_old_contains not in old_text:
raise ValueError("routine expected_old_contains was not found in current routine text")
updated = text[:start] + replacement + text[end:]
status = "replaced"
span = {
"old_line_start": old["line_start"],
"old_line_end": old["line_end"],
"old_sha1": old_sha1,
}
else:
if expected_old_sha1 or expected_old_contains:
raise ValueError("routine old preconditions require an existing routine")
separator = eol + eol if text.strip() else ""
updated = text.rstrip("\r\n") + separator + replacement + eol
status = "appended"
span = {}
return updated, {
"status": status,
"routine": {"kind": new_block["kind"], "name": new_block["name"]},
**span,
}
def strip_line_comment(line: str) -> str:
in_string = False
index = 0
while index < len(line):
char = line[index]
if char == '"':
if in_string and index + 1 < len(line) and line[index + 1] == '"':
index += 2
continue
in_string = not in_string
if not in_string and line[index : index + 2] == "//":
return line[:index]
index += 1
return line
def code_lines(text: str) -> list[str]:
return [strip_line_comment(line) for line in text.replace("\r\n", "\n").replace("\r", "\n").split("\n")]
def validate_bsl_text(text: str) -> dict[str, Any]:
lines = code_lines(text)
clean = "\n".join(lines)
starts = ROUTINE_START_RE.findall(clean)
ends = ROUTINE_END_RE.findall(clean)
region_starts = REGION_START_RE.findall(clean)
region_ends = REGION_END_RE.findall(clean)
preproc_ifs = PREPROC_IF_RE.findall(clean)
preproc_ends = PREPROC_ENDIF_RE.findall(clean)
issues = []
if len(starts) != len(ends):
issues.append(
{
"severity": "error",
"code": "routine_balance",
"message": "Routine start/end count mismatch.",
"starts": len(starts),
"ends": len(ends),
}
)
if len(region_starts) != len(region_ends):
issues.append(
{
"severity": "warning",
"code": "region_balance",
"message": "Region start/end count mismatch.",
"starts": len(region_starts),
"ends": len(region_ends),
}
)
if len(preproc_ifs) != len(preproc_ends):
issues.append(
{
"severity": "warning",
"code": "preprocessor_if_balance",
"message": "Preprocessor #Если/#КонецЕсли count mismatch.",
"starts": len(preproc_ifs),
"ends": len(preproc_ends),
}
)
return {
"schema": "onec_bsl_structural_validation.v1",
"status": "ok" if not any(issue["severity"] == "error" for issue in issues) else "error",
"counts": {
"lines": len(lines),
"routine_starts": len(starts),
"routine_ends": len(ends),
"regions": len(region_starts),
"region_ends": len(region_ends),
"preprocessor_ifs": len(preproc_ifs),
"preprocessor_ends": len(preproc_ends),
},
"routines_sample": [{"kind": kind, "name": name} for kind, name in starts[:80]],
"issues": issues,
}