Files
llm/scripts/plan_1c_task_context.py

384 lines
13 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Build a read-only investigation plan for a textual 1C development task."""
from __future__ import annotations
import argparse
import base64
import json
import re
import sys
from pathlib import Path
from typing import Any
sys.path.insert(0, str(Path(__file__).resolve().parent))
from get_1c_object_brief_context import build_brief_context # noqa: E402
from resolve_1c_object import canonical_kind, load_json, normalize # noqa: E402
from search_1c_object_context import build_search # noqa: E402
DEFAULT_KIND_HINTS = {
"документ": "Document",
"документа": "Document",
"справочник": "Catalog",
"справочника": "Catalog",
"регистр": None,
"регистрсведений": "InformationRegister",
"регистрнакопления": "AccumulationRegister",
"отчет": "Report",
"обработка": "DataProcessor",
"форма": "Form",
}
def decode_arg(value: str | None, encoded: str | None) -> str | None:
if encoded:
return base64.b64decode(encoded).decode("utf-8")
return value
def compact_text(value: str | None) -> str:
return normalize(value or "")
def top_objects(item: dict[str, Any]) -> list[dict[str, Any]]:
return item.get("xml_top_objects") or []
def is_owner_top(top: dict[str, Any]) -> bool:
rel = str(top.get("relative_path") or "")
if "\\Forms\\" in rel or "\\Templates\\" in rel or "\\Commands\\" in rel:
return False
return bool(top.get("name") and top.get("xml_kind"))
def source_from_top(top: dict[str, Any]) -> str:
text = (str(top.get("relative_path") or "") + "\\" + str(top.get("path") or "")).casefold()
return "extension" if "\\расширения\\" in text or "\\extensions\\" in text else "base"
def index_objects(index: dict[str, Any]) -> list[dict[str, Any]]:
result = []
for guid, item in (index.get("objects") or {}).items():
for top in top_objects(item):
if not is_owner_top(top):
continue
kind = canonical_kind(str(top.get("xml_kind") or ""))
result.append(
{
"guid": guid,
"kind": kind,
"name": top.get("name"),
"synonym": top.get("synonym"),
"source": source_from_top(top),
"relative_path": top.get("relative_path"),
"path": top.get("path"),
}
)
return result
def kind_hint(text: str) -> str | None:
text_norm = compact_text(text)
for word, kind in DEFAULT_KIND_HINTS.items():
if compact_text(word) in text_norm:
return kind
return None
def candidate_score(task_norm: str, obj: dict[str, Any], hinted_kind: str | None) -> tuple[int, list[str]]:
score = 0
reasons = []
name_norm = compact_text(obj.get("name"))
synonym_norm = compact_text(obj.get("synonym"))
if name_norm and name_norm in task_norm:
score += 100
reasons.append("name_in_task")
if synonym_norm and synonym_norm in task_norm:
score += 95
reasons.append("synonym_in_task")
if hinted_kind and obj.get("kind") == hinted_kind:
score += 55
reasons.append("kind_hint")
elif hinted_kind and obj.get("kind") != hinted_kind:
score -= 70
reasons.append("kind_mismatch")
if obj.get("source") == "base":
score += 12
else:
score -= 8
return score, reasons
def find_object_candidates(index: dict[str, Any], text: str, limit: int) -> list[dict[str, Any]]:
task_norm = compact_text(text)
hinted_kind = kind_hint(text)
by_key: dict[tuple[str | None, str], dict[str, Any]] = {}
for obj in index_objects(index):
score, reasons = candidate_score(task_norm, obj, hinted_kind)
if score <= 0:
continue
key = (obj.get("kind"), compact_text(obj.get("name")))
candidate = {**obj, "score": score, "reasons": reasons}
previous = by_key.get(key)
if previous is None:
by_key[key] = candidate
continue
previous_rank = (previous.get("source") == "base", previous.get("score", 0))
candidate_rank = (candidate.get("source") == "base", candidate.get("score", 0))
if candidate_rank > previous_rank:
by_key[key] = candidate
candidates = list(by_key.values())
candidates.sort(key=lambda row: (-row["score"], row.get("source") != "base", row.get("kind") or "", row.get("name") or ""))
return candidates[:limit]
def quoted_phrases(text: str) -> list[str]:
result = []
for pattern in (r'"([^"]{2,80})"', r"'([^']{2,80})'", r"«([^»]{2,80})»"):
result.extend(match.group(1).strip() for match in re.finditer(pattern, text))
return result
def identifier_terms(text: str) -> list[str]:
raw = re.findall(r"[A-Za-zА-Яа-яЁё_][A-Za-zА-Яа-яЁё0-9_]{3,}", text)
stop = {
"нужно",
"надо",
"добавить",
"сделать",
"проверить",
"показать",
"изменить",
"чтобы",
"если",
"тогда",
"логика",
"логику",
"кнопка",
"кнопку",
"форма",
"форму",
"документе",
"документ",
"документа",
"справочник",
"реквизит",
"реквизиты",
}
result = []
for item in raw:
if item.casefold() in stop:
continue
if item not in result:
result.append(item)
return result
def search_terms(text: str, object_candidates: list[dict[str, Any]], limit: int) -> list[str]:
terms = quoted_phrases(text) + identifier_terms(text)
object_names = {compact_text(row.get("name")) for row in object_candidates}
object_synonyms = {compact_text(row.get("synonym")) for row in object_candidates}
result = []
for term in terms:
norm = compact_text(term)
if not norm or norm in object_names or norm in object_synonyms:
continue
if term not in result:
result.append(term)
if text not in result:
result.append(text)
return result[:limit]
def short_matches(search_result: dict[str, Any], max_matches: int) -> list[dict[str, Any]]:
rows = []
for match in search_result.get("matches") or []:
row = {
"area": match.get("area"),
"name": match.get("name"),
"title": match.get("title"),
"origin": match.get("origin"),
"effective_action": match.get("effective_action"),
"form": match.get("form"),
"evidence": match.get("evidence"),
}
rows.append({key: value for key, value in row.items() if value not in (None, [], "")})
if len(rows) >= max_matches:
break
return rows
def object_investigation(
index: dict[str, Any],
candidate: dict[str, Any],
*,
view: str,
search_texts: list[str],
max_matches: int,
) -> dict[str, Any]:
kind = candidate.get("kind")
name = candidate.get("name")
brief = build_brief_context(
index,
kind=kind,
name=name,
view=view,
extension=None,
max_attributes=25,
max_modules=20,
max_forms=20,
)
searches = []
for text in search_texts:
search = build_search(
index,
kind=kind,
name=name,
query=text,
view=view,
extension=None,
search_code=True,
max_form_items=1200,
limit=max_matches,
)
if search.get("matches"):
searches.append(
{
"text": text,
"match_count": len(search.get("matches") or []),
"matches": short_matches(search, max_matches=max_matches),
}
)
return {
"candidate": candidate,
"brief": {
"object": brief.get("object"),
"active_extensions": brief.get("active_extensions"),
"counts": brief.get("counts"),
"attributes_preview": brief.get("attributes", {}).get("items", [])[:10],
"forms": brief.get("forms", {}).get("items", []),
"modules": brief.get("modules", {}).get("items", []),
},
"searches": searches,
"recommended_reads": build_recommended_reads(kind, name, brief, searches),
}
def build_recommended_reads(kind: str, name: str, brief: dict[str, Any], searches: list[dict[str, Any]]) -> list[dict[str, Any]]:
reads = [
{
"purpose": "full_effective_metadata",
"command": f"python scripts/get_1c_object_metadata.py --kind {kind} --name {name} --view effective",
}
]
seen_forms = set()
seen_modules = set()
for search in searches:
for match in search.get("matches") or []:
form = match.get("form")
if form and form not in seen_forms:
seen_forms.add(form)
reads.append(
{
"purpose": "inspect_form",
"form": form,
"command": f"python scripts/get_1c_form_context.py --kind {kind} --name {name} --form {form} --view effective",
}
)
area = match.get("area")
module_name = match.get("name")
if area in {"module", "module.code"} and module_name and module_name not in seen_modules:
seen_modules.add(module_name)
reads.append(
{
"purpose": "inspect_module",
"module": module_name,
"command": f"python scripts/get_1c_module.py --kind {kind} --name {name} --module {module_name} --view effective",
}
)
if len(reads) == 1:
for module in (brief.get("modules") or {}).get("items", []):
module_name = module.get("name")
if module_name in {"МодульОбъекта", "МодульМенеджера"}:
reads.append(
{
"purpose": "inspect_likely_module",
"module": module_name,
"command": f"python scripts/get_1c_module.py --kind {kind} --name {name} --module {module_name} --view effective --max-chars 12000",
}
)
return reads[:10]
def build_plan(
index: dict[str, Any],
*,
text: str,
view: str,
max_objects: int,
max_terms: int,
max_matches: int,
) -> dict[str, Any]:
candidates = find_object_candidates(index, text, limit=max_objects)
terms = search_terms(text, candidates, limit=max_terms)
investigations = [
object_investigation(index, candidate, view=view, search_texts=terms, max_matches=max_matches)
for candidate in candidates
]
return {
"schema": "onec_task_context_plan.v1",
"view": view,
"task": {"text": text},
"object_candidates": candidates,
"search_terms": terms,
"investigations": investigations,
"safety": {
"mode": "read_only",
"write_status": "blocked_until_write_gates",
"write_contract": "docs/1c-write-path-safety.md",
},
"counts": {
"object_candidates": len(candidates),
"search_terms": len(terms),
"investigations": len(investigations),
},
}
def main() -> int:
parser = argparse.ArgumentParser(description="Plan read-only 1C task context gathering.")
parser.add_argument("--index", type=Path, required=True)
parser.add_argument("--text")
parser.add_argument("--text-b64")
parser.add_argument("--view", choices=["effective", "base"], default="effective")
parser.add_argument("--max-objects", type=int, default=5)
parser.add_argument("--max-terms", type=int, default=8)
parser.add_argument("--max-matches", type=int, default=8)
parser.add_argument("--output", type=Path)
args = parser.parse_args()
text = decode_arg(args.text, args.text_b64)
if not text:
raise SystemExit("Use --text or --text-b64.")
result = build_plan(
load_json(args.index),
text=text,
view=args.view,
max_objects=args.max_objects,
max_terms=args.max_terms,
max_matches=args.max_matches,
)
output = json.dumps(result, ensure_ascii=False, indent=2) + "\n"
if args.output:
args.output.parent.mkdir(parents=True, exist_ok=True)
args.output.write_text(output, encoding="utf-8")
print(json.dumps({"output": str(args.output), "counts": result["counts"], "view": result["view"]}, ensure_ascii=False))
else:
print(output)
return 0
if __name__ == "__main__":
raise SystemExit(main())