Initial SQL-only 1C adapter baseline

This commit is contained in:
2026-07-22 03:03:47 +03:00
commit e2503b77e7
545 changed files with 184711 additions and 0 deletions
+174
View File
@@ -0,0 +1,174 @@
from __future__ import annotations
import argparse
import json
import sqlite3
from pathlib import Path
from typing import Any
from common import cosine_similarity, read_jsonl, corpus_content_hash, unpack_float_vector
from rag_embedding_providers import LOCAL_HASHING_PROVIDER, embed_texts
from rag_profiles import resolve_rag_profile
ROOT = Path(__file__).resolve().parents[1]
DEFAULT_CORPUS = ROOT / "plugins" / "1c" / "datasets" / "prepared" / "rag_corpus.jsonl"
DEFAULT_INDEX = ROOT / "plugins" / "1c" / "datasets" / "prepared" / "rag_vector_index.sqlite"
def load_meta(conn: sqlite3.Connection) -> dict[str, Any]:
meta: dict[str, Any] = {}
for key, value in conn.execute("SELECT key, value FROM vector_meta"):
try:
meta[str(key)] = json.loads(value)
except json.JSONDecodeError:
meta[str(key)] = value
return meta
def freshness(index_meta: dict[str, Any], corpus_path: Path | None) -> dict[str, Any]:
if not corpus_path:
return {"status": "unknown", "reason": "corpus path was not provided"}
if not corpus_path.exists():
return {"status": "unknown", "reason": "corpus file is missing", "corpus": str(corpus_path)}
current_hash = corpus_content_hash(read_jsonl(corpus_path))
index_hash = str(index_meta.get("corpus_hash") or "")
return {
"status": "fresh" if current_hash == index_hash else "stale",
"corpus": str(corpus_path),
"corpus_hash": current_hash,
"index_corpus_hash": index_hash,
}
def row_document(row: sqlite3.Row) -> dict[str, Any]:
try:
metadata = json.loads(row["metadata_json"] or "{}")
except json.JSONDecodeError:
metadata = {}
return {
"id": row["id"],
"document_id": row["document_id"],
"source_path": row["source_path"],
"title": row["title"],
"chunk_index": row["chunk_index"],
"content": row["content"],
"metadata": metadata,
}
def search_vector_index(
index_path: Path,
query: str,
*,
limit: int,
candidate_limit: int | None = None,
min_score: float = 0.0,
source_types: list[str] | None = None,
corpus_path: Path | None = DEFAULT_CORPUS,
embedding_base_url: str = "",
embedding_api_key_env: str = "OPENAI_API_KEY",
) -> dict[str, Any]:
conn = sqlite3.connect(index_path)
conn.row_factory = sqlite3.Row
try:
meta = load_meta(conn)
dimensions = int(meta.get("embedding_dimensions") or 0)
if dimensions <= 0:
raise ValueError("Vector index metadata is missing embedding_dimensions")
provider = str(meta.get("embedding_provider") or LOCAL_HASHING_PROVIDER)
model = str(meta.get("embedding_model") or "")
query_vector = embed_texts(
[query],
provider=provider,
model=model,
dimensions=dimensions,
base_url=embedding_base_url or str(meta.get("embedding_base_url") or ""),
api_key_env=embedding_api_key_env,
)[0]
if len(query_vector) != dimensions:
raise ValueError(f"Query embedding dimensions {len(query_vector)} do not match index dimensions {dimensions}")
allowed_source_types = {value.lower() for value in source_types or []}
sql = "SELECT * FROM vector_documents"
params: list[Any] = []
if allowed_source_types:
placeholders = ",".join("?" for _ in allowed_source_types)
sql += f" WHERE lower(source_type) IN ({placeholders})"
params.extend(sorted(allowed_source_types))
results = []
for row in conn.execute(sql, params):
score = cosine_similarity(query_vector, unpack_float_vector(row["vector"], dimensions))
if score > min_score:
results.append({"score": score, "document": row_document(row)})
results.sort(key=lambda item: item["score"], reverse=True)
if candidate_limit:
results = results[:candidate_limit]
return {
"schema": "onec_rag_vector_search.v1",
"status": "ok",
"query": query,
"index": str(index_path),
"meta": meta,
"freshness": freshness(meta, corpus_path),
"results": results[:limit],
"counts": {"results": min(len(results), limit), "candidates": len(results)},
}
finally:
conn.close()
def print_result(result: dict, index: int) -> None:
document = result["document"]
content = (document.get("content") or "").strip().replace("\n", " ")
if len(content) > 500:
content = content[:497].rstrip() + "..."
print(f"{index}. score={result['score']:.4f}")
print(f" source={document.get('source_path')} chunk={document.get('chunk_index')}")
print(f" title={document.get('title')}")
if (document.get("metadata") or {}).get("heading"):
print(f" heading={(document.get('metadata') or {}).get('heading')}")
print(f" content={content}")
def main() -> int:
parser = argparse.ArgumentParser(description="Search the local 1C RAG SQLite vector index.")
parser.add_argument("query")
parser.add_argument("--index", type=Path, default=DEFAULT_INDEX)
parser.add_argument("--corpus", type=Path, default=DEFAULT_CORPUS)
parser.add_argument("--profile", default="auto")
parser.add_argument("--limit", type=int)
parser.add_argument("--candidate-limit", type=int)
parser.add_argument("--min-score", type=float, default=0.0)
parser.add_argument("--source-type", action="append", dest="source_types")
parser.add_argument("--embedding-base-url", default="")
parser.add_argument("--embedding-api-key-env", default="OPENAI_API_KEY")
parser.add_argument("--json", action="store_true")
args = parser.parse_args()
profile = resolve_rag_profile(args.profile, args.query)
source_types = args.source_types if args.source_types is not None else profile["source_types"]
result = search_vector_index(
args.index,
args.query,
limit=args.limit or int(profile["limit"]),
candidate_limit=args.candidate_limit or int(profile["candidate_limit"]),
min_score=float(args.min_score),
source_types=source_types,
corpus_path=args.corpus,
embedding_base_url=args.embedding_base_url,
embedding_api_key_env=args.embedding_api_key_env,
)
if args.json:
print(json.dumps(result, ensure_ascii=False, indent=2))
elif result["results"]:
freshness_status = (result.get("freshness") or {}).get("status")
print(f"freshness={freshness_status} embedding_model={result['meta'].get('embedding_model')}")
for position, item in enumerate(result["results"], start=1):
print_result(item, position)
else:
print("No matches.")
return 0 if (result.get("freshness") or {}).get("status") != "stale" else 1
if __name__ == "__main__":
raise SystemExit(main())