Initial SQL-only 1C adapter baseline

This commit is contained in:
2026-07-22 03:03:47 +03:00
commit e2503b77e7
545 changed files with 184711 additions and 0 deletions
+289
View File
@@ -0,0 +1,289 @@
from __future__ import annotations
import argparse
import hashlib
import json
import re
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
DEFAULT_SOURCE_DIR = ROOT / "plugins" / "1c" / "rag" / "sources"
DEFAULT_OUTPUT = ROOT / "plugins" / "1c" / "datasets" / "prepared" / "rag_corpus.jsonl"
SUPPORTED_EXTENSIONS = {".md", ".txt", ".bsl", ".os"}
HEADING_RE = re.compile(r"^(#{1,6})\s+(.+)$")
FRONT_MATTER_RE = re.compile(r"^---\n(?P<body>.*?)\n---\n?", re.DOTALL)
def normalize_text(text: str) -> str:
lines = [line.rstrip() for line in text.replace("\r\n", "\n").split("\n")]
return "\n".join(lines).strip()
def parse_front_matter(text: str) -> tuple[dict[str, str], str]:
match = FRONT_MATTER_RE.match(text)
if not match:
return {}, text
metadata: dict[str, str] = {}
for line in match.group("body").splitlines():
if ":" not in line:
continue
key, value = line.split(":", 1)
value = value.strip().strip('"').strip("'")
metadata[key.strip()] = value
return metadata, text[match.end() :].strip()
def split_paragraphs(text: str) -> list[str]:
blocks: list[str] = []
current: list[str] = []
for line in text.split("\n"):
stripped = line.strip()
if not stripped:
if current:
blocks.append("\n".join(current).strip())
current = []
continue
if HEADING_RE.match(stripped) and current:
blocks.append("\n".join(current).strip())
current = [stripped]
else:
current.append(line.rstrip())
if current:
blocks.append("\n".join(current).strip())
return [block for block in blocks if block]
def hard_split_text(text: str, chunk_size: int, overlap: int) -> list[str]:
chunks: list[str] = []
start = 0
while start < len(text):
end = min(start + chunk_size, len(text))
chunk = text[start:end].strip()
if chunk:
chunks.append(chunk)
if end == len(text):
break
start = end - overlap
return chunks
def chunk_text(text: str, chunk_size: int, overlap: int) -> list[dict]:
if chunk_size <= 0:
raise ValueError("chunk_size must be positive")
if overlap < 0:
raise ValueError("overlap must not be negative")
if overlap >= chunk_size:
raise ValueError("overlap must be smaller than chunk_size")
chunks: list[dict] = []
current: list[str] = []
current_heading = ""
last_tail = ""
headings_seen: list[str] = []
def flush() -> None:
nonlocal current, last_tail
if not current:
return
content = "\n\n".join(current).strip()
if last_tail and not content.startswith(last_tail):
content = f"{last_tail}\n\n{content}".strip()
chunks.append(
{
"content": content,
"heading": current_heading,
"headings": headings_seen[-4:],
}
)
last_tail = content[-overlap:].strip() if overlap else ""
current = []
for block in split_paragraphs(text):
heading_match = HEADING_RE.match(block.split("\n", 1)[0].strip())
if heading_match:
current_heading = heading_match.group(2).strip()
headings_seen.append(current_heading)
if len(block) > chunk_size:
flush()
for part in hard_split_text(block, chunk_size, overlap):
chunks.append(
{
"content": part,
"heading": current_heading,
"headings": headings_seen[-4:],
}
)
last_tail = part[-overlap:].strip() if overlap else ""
continue
candidate = "\n\n".join([*current, block]).strip()
if current and len(candidate) > chunk_size:
flush()
current.append(block)
flush()
return chunks
def stable_id(*parts: str) -> str:
digest = hashlib.sha256("|".join(parts).encode("utf-8")).hexdigest()
return digest[:16]
def classify_source(path: Path, text: str, front_matter: dict[str, str] | None = None) -> str:
front_matter = front_matter or {}
if front_matter.get("source") == "official_1c_its":
return front_matter.get("source_type") or "official_1c_its"
name = path.name.lower()
lowered = text[:4000].lower()
if path.suffix.lower() in {".bsl", ".os"}:
return "bsl"
if "1c bsl module snapshot" in lowered or "```bsl" in lowered or "процедура " in lowered:
return "bsl"
if "1c metadata snapshot" in lowered or "metadata snapshot" in lowered:
return "metadata"
if "read-only" in lowered or "только чтение" in lowered or "выбрать" in lowered and "запрос" in lowered:
return "query"
if "production" in lowered or "резервн" in lowered or "согласован" in lowered or "опасн" in lowered:
return "safety"
if "metadata" in name:
return "metadata"
if "bsl" in name or "module" in name:
return "bsl"
return "docs"
def iter_source_files(source_dir: Path) -> list[Path]:
if not source_dir.exists():
return []
return sorted(
path
for path in source_dir.rglob("*")
if path.is_file() and path.suffix.lower() in SUPPORTED_EXTENSIONS
)
def build_records(source_dir: Path, chunk_size: int, overlap: int) -> list[dict]:
records: list[dict] = []
for path in iter_source_files(source_dir):
relative_path = path.relative_to(source_dir).as_posix()
text = normalize_text(path.read_text(encoding="utf-8"))
if not text:
continue
front_matter, body = parse_front_matter(text)
chunk_source_text = body or text
source_type = classify_source(path, chunk_source_text, front_matter)
chunks = chunk_text(chunk_source_text, chunk_size=chunk_size, overlap=overlap)
document_id = stable_id(relative_path, chunk_source_text)
base_title = front_matter.get("title") or path.stem.replace("_", " ").replace("-", " ")
for index, chunk in enumerate(chunks):
content = chunk["content"]
title = chunk.get("heading") or base_title
records.append(
{
"id": stable_id(document_id, str(index), content),
"document_id": document_id,
"source_path": relative_path,
"title": title,
"chunk_index": index,
"content": content,
"metadata": {
"domain": "1c",
"source_type": source_type,
"file_type": path.suffix.lower().lstrip("."),
"access": front_matter.get("access") or "",
"official_source": front_matter.get("source") or "",
"url": front_matter.get("url") or "",
"platform_family": front_matter.get("platform_family") or "",
"platform_version": front_matter.get("platform_version") or "",
"platform_doc_id": front_matter.get("platform_doc_id") or "",
"doc_book": front_matter.get("doc_book") or "",
"doc_bookmark": front_matter.get("doc_bookmark") or "",
"doc_content_id": front_matter.get("doc_content_id") or "",
"doc_coordinate": front_matter.get("doc_coordinate") or "",
"access_blocked": front_matter.get("access_blocked") or "",
"access_findings": front_matter.get("access_findings") or "",
"source_sha256": front_matter.get("source_sha256") or "",
"chunk_size": chunk_size,
"overlap": overlap,
"heading": chunk.get("heading") or "",
"headings": chunk.get("headings") or [],
},
}
)
return records
def build_manifest(source_dir: Path, records: list[dict], chunk_size: int, overlap: int) -> dict:
by_source: dict[str, dict] = {}
for record in records:
source_path = record["source_path"]
source = by_source.setdefault(
source_path,
{
"source_path": source_path,
"title": record.get("title"),
"source_type": (record.get("metadata") or {}).get("source_type"),
"file_type": (record.get("metadata") or {}).get("file_type"),
"document_id": record.get("document_id"),
"chunk_count": 0,
"content_hash": "",
},
)
source["chunk_count"] += 1
for relative_path, source in by_source.items():
path = source_dir / relative_path
text = normalize_text(path.read_text(encoding="utf-8")) if path.exists() else ""
source["content_hash"] = hashlib.sha256(text.encode("utf-8")).hexdigest()
return {
"schema_version": 1,
"source_dir": str(source_dir),
"chunk_size": chunk_size,
"overlap": overlap,
"source_count": len(by_source),
"chunk_count": len(records),
"sources": sorted(by_source.values(), key=lambda item: item["source_path"]),
}
def write_jsonl(path: Path, records: list[dict]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
with path.open("w", encoding="utf-8") as handle:
for record in records:
handle.write(json.dumps(record, ensure_ascii=False) + "\n")
def main() -> int:
parser = argparse.ArgumentParser(description="Prepare a JSONL RAG corpus for the 1C plugin.")
parser.add_argument("--source-dir", type=Path, default=DEFAULT_SOURCE_DIR)
parser.add_argument("--output", type=Path, default=DEFAULT_OUTPUT)
parser.add_argument("--manifest", type=Path)
parser.add_argument("--chunk-size", type=int, default=1800)
parser.add_argument("--overlap", type=int, default=200)
args = parser.parse_args()
records = build_records(
source_dir=args.source_dir,
chunk_size=args.chunk_size,
overlap=args.overlap,
)
write_jsonl(args.output, records)
manifest_path = args.manifest or args.output.with_suffix(".manifest.json")
manifest_path.parent.mkdir(parents=True, exist_ok=True)
manifest_path.write_text(
json.dumps(build_manifest(args.source_dir, records, args.chunk_size, args.overlap), ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
print(f"Wrote {len(records)} chunk(s) to {args.output}")
print(f"Wrote RAG manifest to {manifest_path}")
return 0
if __name__ == "__main__":
raise SystemExit(main())