Files
llm/scripts/normalize_1c_its_docs.py
T

674 lines
24 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
from __future__ import annotations
import argparse
import hashlib
import html
import json
import re
import urllib.parse
from html.parser import HTMLParser
from pathlib import Path
from typing import Any
from one_c_its_platform import merge_platform_metadata
ROOT = Path(__file__).resolve().parents[1]
DEFAULT_RAW_DIR = ROOT / "plugins" / "1c" / "rag" / "official-docs" / "raw"
DEFAULT_RAW_MANIFEST = DEFAULT_RAW_DIR / "manifest.json"
DEFAULT_OUTPUT_DIR = ROOT / "plugins" / "1c" / "rag" / "official-docs" / "normalized"
DEFAULT_RAG_SOURCE_DIR = ROOT / "plugins" / "1c" / "rag" / "sources" / "official" / "its"
DEFAULT_MANIFEST_OUTPUT = DEFAULT_OUTPUT_DIR / "manifest.json"
BLOCK_TAGS = {
"address",
"article",
"aside",
"blockquote",
"br",
"div",
"dl",
"fieldset",
"figcaption",
"figure",
"footer",
"form",
"h1",
"h2",
"h3",
"h4",
"h5",
"h6",
"header",
"hr",
"li",
"main",
"nav",
"ol",
"p",
"pre",
"section",
"table",
"td",
"th",
"tr",
"ul",
}
BOILERPLATE_LINES = {
"Вход",
"Об 1С:ИТС",
"Тест-драйв",
"Заказать ИТС",
"Задать вопрос",
"Обновить ПО",
"Оценить 1С",
"Купить кассу",
"Тематические подборки",
"Календарь бухгалтера",
"Калькуляторы",
"Подбор КБК",
"Последние результаты поиска",
"Подписаться на рассылку",
"Главная",
"Инструкции по разработке на 1С",
"Платформа 1С:Предприятие. Документация",
"Содержание",
"Результаты поиска",
"Вконтакте",
"Принимаю",
"Методические материалы для разработчиков и администраторов 1С",
"Глоссарий разработчика",
"Назад",
}
BOILERPLATE_PREFIXES = (
"© Фирма «1С»",
"Информационная система 1С:ИТС",
"Инструкции по учету в программах 1С",
"Новости1С:Лекторий",
"Мы используем файлы cookie",
"Продолжая находиться на сайте",
"на условиях, указанных по ссылке",
)
NAVIGATION_ONLY_TERMS = {
"Руководство разработчика",
"Руководство администратора",
"Клиент-серверный вариант. Руководство администратора",
"Руководство разработчика. Обычный режим",
"Руководство пользователя",
}
VERSION_SELECTOR_LINE_RE = re.compile(r"^(?:\d+\.\d+(?:\.\d+)?){4,}$")
def charset_from_content_type(content_type: str) -> str | None:
match = re.search(r"charset=([^;\s]+)", content_type, flags=re.IGNORECASE)
if match:
return match.group(1).strip("\"'")
return None
def charset_from_html_head(raw_bytes: bytes) -> str | None:
head = raw_bytes[:4096].decode("ascii", errors="ignore")
match = re.search(r"charset\s*=\s*['\"]?([^'\"\s/>;]+)", head, flags=re.IGNORECASE)
if match:
return match.group(1).strip()
return None
def decode_html(raw_bytes: bytes, record: dict[str, Any]) -> str:
candidates = [
charset_from_content_type(str(record.get("content_type") or "")),
charset_from_html_head(raw_bytes),
"utf-8",
"cp1251",
]
seen: set[str] = set()
for charset in candidates:
if not charset:
continue
key = charset.lower()
if key in seen:
continue
seen.add(key)
try:
return raw_bytes.decode(charset)
except (LookupError, UnicodeDecodeError):
continue
return raw_bytes.decode("utf-8", errors="replace")
class TextExtractor(HTMLParser):
def __init__(self) -> None:
super().__init__(convert_charrefs=True)
self.parts: list[str] = []
self.skip_stack: list[str] = []
self.heading_stack: list[str] = []
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
tag = tag.lower()
if tag in {"head", "script", "style", "noscript", "svg"}:
self.skip_stack.append(tag)
return
if self.skip_stack:
return
if tag in {"h1", "h2", "h3", "h4", "h5", "h6"}:
self.heading_stack.append(tag)
self.parts.append("\n\n" + "#" * int(tag[1]) + " ")
elif tag == "li":
self.parts.append("\n- ")
elif tag == "br":
self.parts.append("\n")
elif tag in BLOCK_TAGS:
self.parts.append("\n\n")
def handle_endtag(self, tag: str) -> None:
tag = tag.lower()
if self.skip_stack and self.skip_stack[-1] == tag:
self.skip_stack.pop()
return
if self.skip_stack:
return
if tag in BLOCK_TAGS or tag in self.heading_stack:
self.parts.append("\n\n")
if self.heading_stack and self.heading_stack[-1] == tag:
self.heading_stack.pop()
def handle_data(self, data: str) -> None:
if self.skip_stack:
return
text = html.unescape(data)
if text.strip():
self.parts.append(text)
def text(self) -> str:
raw = "".join(self.parts)
raw = raw.replace("\r\n", "\n").replace("\r", "\n")
raw = re.sub(r"[ \t]+", " ", raw)
raw = re.sub(r"\n{3,}", "\n\n", raw)
lines = [line.strip() for line in raw.split("\n")]
return "\n".join(line for line in lines).strip()
class LinkExtractor(HTMLParser):
def __init__(self) -> None:
super().__init__(convert_charrefs=True)
self.links: list[str] = []
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
tag = tag.lower()
attr_name = "href" if tag == "a" else "src" if tag in {"iframe", "frame"} else None
if attr_name is None:
return
for name, value in attrs:
if name.lower() == attr_name and value:
self.links.append(value)
class MediaExtractor(HTMLParser):
def __init__(self, base_url: str) -> None:
super().__init__(convert_charrefs=True)
self.base_url = base_url
self.images: list[dict[str, Any]] = []
self.table_count = 0
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
tag = tag.lower()
values = {name.lower(): value for name, value in attrs if value is not None}
if tag == "table":
self.table_count += 1
if tag != "img":
return
src = values.get("src") or ""
url = normalize_url(self.base_url, src)
if not url or is_tracking_image_url(url):
return
self.images.append(
{
"url": url,
"src": src,
"alt": values.get("alt") or "",
"title": values.get("title") or "",
"width": values.get("width") or "",
"height": values.get("height") or "",
}
)
def load_json(path: Path) -> dict[str, Any]:
data = json.loads(path.read_text(encoding="utf-8-sig"))
if not isinstance(data, dict):
raise ValueError(f"{path} must contain a JSON object")
return data
def normalize_url(base_url: str, href: str) -> str | None:
if href.startswith(("mailto:", "tel:", "javascript:")):
return None
absolute = urllib.parse.urljoin(base_url, href)
parsed = urllib.parse.urlparse(absolute)
if parsed.scheme not in {"http", "https"}:
return None
parsed = parsed._replace(fragment="")
return urllib.parse.urlunparse(parsed)
def is_content_src_url(url: str) -> bool:
path = urllib.parse.urlparse(url).path
return "/db/content/" in path and "/src/" in path
def is_tracking_image_url(url: str) -> bool:
lowered = url.lower()
return any(
marker in lowered
for marker in (
"mc.yandex",
"counter.yadro",
"logo_its",
"favicon",
"apple_touch",
"/watch/",
)
)
def raw_slug_for_url(value: str) -> str:
parsed = urllib.parse.urlparse(value)
raw = f"{parsed.netloc}{parsed.path}".strip("/") or "index"
if parsed.query:
raw += "_" + parsed.query
slug = re.sub(r"[^A-Za-zА-Яа-яЁё0-9_.-]+", "_", raw, flags=re.UNICODE).strip("_")
digest = hashlib.sha256(value.encode("utf-8")).hexdigest()[:12]
return f"{slug[:120]}__{digest}.html"
def strip_list_marker(line: str) -> str:
return re.sub(r"^[-*]\s+", "", line.strip()).strip()
def is_boilerplate_line(line: str) -> bool:
normalized = strip_list_marker(line)
if not normalized or normalized == "-":
return True
compact = normalized.replace(" ", "")
if VERSION_SELECTOR_LINE_RE.match(compact):
return True
if normalized in BOILERPLATE_LINES:
return True
if normalized in NAVIGATION_ONLY_TERMS:
return True
return any(normalized.startswith(prefix) for prefix in BOILERPLATE_PREFIXES)
def compact_lines(lines: list[str]) -> str:
compact = []
previous_blank = False
for line in lines:
blank = not line
if blank and previous_blank:
continue
compact.append(line)
previous_blank = blank
cleaned = "\n".join(compact).strip()
cleaned = re.sub(r"\n{3,}", "\n\n", cleaned)
return cleaned
def clean_glossary_text(text: str, title: str) -> str | None:
term = title.split("::", 1)[0].strip()
raw_lines = [line.strip() for line in text.splitlines()]
start_index = None
for index, line in enumerate(raw_lines):
plain = strip_list_marker(line).lstrip("#").strip()
if plain == term:
start_index = index + 1
break
if start_index is None:
return None
body = []
for line in raw_lines[start_index:]:
plain = strip_list_marker(line).lstrip("#").strip()
if plain == "Назад":
break
if plain in {term, title}:
continue
if is_boilerplate_line(line):
continue
body.append(line)
return compact_lines(body)
def clean_its_text(text: str, title: str, source_type: str | None = None) -> str:
if source_type == "official_1c_its_glossary":
glossary_text = clean_glossary_text(text, title)
if glossary_text is not None:
return glossary_text
term = title.split("::", 1)[0].strip()
fallback_lines = []
for raw_line in text.splitlines():
line = raw_line.strip()
plain = strip_list_marker(line).lstrip("#").strip()
if plain in {term, title}:
continue
if is_boilerplate_line(line):
continue
fallback_lines.append(line)
return compact_lines(fallback_lines)
lines = []
for raw_line in text.splitlines():
line = raw_line.strip()
if is_boilerplate_line(line):
continue
# Keep real headings, but drop duplicate title-only body lines.
if strip_list_marker(line) == title:
continue
lines.append(line)
return compact_lines(lines)
def content_quality(text: str, title: str) -> dict[str, Any]:
lines = [line.strip() for line in text.splitlines() if line.strip()]
content_lines = []
for line in lines:
plain = strip_list_marker(line).lstrip("#").strip()
if not plain or plain == title:
continue
if is_boilerplate_line(plain):
continue
content_lines.append(plain)
prose_lines = [
line
for line in content_lines
if len(line) >= 45
and not line.startswith("#")
and not line.startswith("- ")
and not line.endswith(":")
]
word_count = sum(len(re.findall(r"[A-Za-zА-Яа-яЁё0-9_]+", line)) for line in content_lines)
return {
"content_lines": len(content_lines),
"prose_lines": len(prose_lines),
"word_count": word_count,
"chars": len(text),
"is_content": len(prose_lines) >= 1 and word_count >= 25,
}
def safe_name(value: str, fallback: str) -> str:
name = re.sub(r"[^A-Za-zА-Яа-яЁё0-9_.-]+", "_", value, flags=re.UNICODE).strip("_")
return (name or fallback)[:120]
def front_matter(record: dict[str, Any], source_hash: str) -> str:
record = merge_platform_metadata(record, str(record.get("url") or ""))
fields = {
"source": "official_1c_its",
"access": "licensed_private",
"source_id": record.get("source_id"),
"source_type": record.get("source_type"),
"title": record.get("title"),
"url": record.get("url"),
"platform_family": record.get("platform_family"),
"platform_version": record.get("platform_version"),
"platform_doc_id": record.get("platform_doc_id"),
"doc_book": record.get("doc_book"),
"doc_bookmark": record.get("doc_bookmark"),
"doc_content_id": record.get("doc_content_id"),
"doc_coordinate": record.get("doc_coordinate"),
"access_blocked": "true" if record.get("access_blocked") else "",
"access_findings": ",".join(str(item) for item in (record.get("access_findings") or [])),
"source_sha256": source_hash,
}
lines = ["---"]
for key, value in fields.items():
if value in (None, ""):
continue
escaped = str(value).replace('"', '\\"')
lines.append(f'{key}: "{escaped}"')
lines.append("---")
return "\n".join(lines)
def media_markdown(media: dict[str, Any]) -> str:
images = media.get("images") or []
if not images:
return ""
lines = ["## Иллюстрации", ""]
for index, image in enumerate(images, start=1):
alt = str(image.get("alt") or image.get("title") or f"Иллюстрация {index}").strip()
url = str(image.get("url") or "").strip()
size = "x".join(part for part in [str(image.get("width") or ""), str(image.get("height") or "")] if part)
suffix = f" ({size})" if size else ""
lines.append(f"- ![{alt}]({url}){suffix}")
return "\n".join(lines).strip() + "\n"
def normalize_page(record: dict[str, Any], *, raw_dir: Path, output_dir: Path, rag_source_dir: Path) -> dict[str, Any] | None:
record = merge_platform_metadata(record, str(record.get("url") or ""))
raw_path = raw_dir / str(record.get("file") or "")
if not raw_path.exists():
return None
raw_bytes = raw_path.read_bytes()
raw_text = decode_html(raw_bytes, record)
extractor = TextExtractor()
extractor.feed(raw_text)
media_extractor = MediaExtractor(str(record.get("url") or ""))
media_extractor.feed(raw_text)
media = {"images": media_extractor.images, "table_count": media_extractor.table_count}
title = str(record.get("title") or record.get("source_title") or record.get("source_id") or "its_doc")
text = clean_its_text(extractor.text(), title, str(record.get("source_type") or ""))
if not text:
return {
"source_id": record.get("source_id"),
"source_type": record.get("source_type"),
"access": "licensed_private",
"title": title,
"url": record.get("url"),
"platform_family": record.get("platform_family"),
"platform_version": record.get("platform_version"),
"platform_doc_id": record.get("platform_doc_id"),
"doc_book": record.get("doc_book"),
"doc_bookmark": record.get("doc_bookmark"),
"doc_content_id": record.get("doc_content_id"),
"doc_coordinate": record.get("doc_coordinate"),
"access_blocked": record.get("access_blocked"),
"access_findings": record.get("access_findings") or [],
"raw_file": record.get("file"),
"skipped": True,
"skip_reason": "empty_after_cleaning",
"quality": content_quality("", title),
"media": media,
}
quality = content_quality(text, title)
if not quality["is_content"]:
return {
"source_id": record.get("source_id"),
"source_type": record.get("source_type"),
"access": "licensed_private",
"title": title,
"url": record.get("url"),
"platform_family": record.get("platform_family"),
"platform_version": record.get("platform_version"),
"platform_doc_id": record.get("platform_doc_id"),
"doc_book": record.get("doc_book"),
"doc_bookmark": record.get("doc_bookmark"),
"doc_content_id": record.get("doc_content_id"),
"doc_coordinate": record.get("doc_coordinate"),
"access_blocked": record.get("access_blocked"),
"access_findings": record.get("access_findings") or [],
"raw_file": record.get("file"),
"skipped": True,
"skip_reason": "navigation_or_low_content",
"quality": quality,
"media": media,
}
source_hash = hashlib.sha256(raw_bytes).hexdigest()
stem = safe_name(f"{record.get('source_id')}_{title}", fallback=source_hash[:12])
file_name = f"{stem}__{source_hash[:12]}.md"
media_section = media_markdown(media)
content = f"{front_matter(record, source_hash)}\n\n# {title}\n\n{text}\n"
if media_section:
content += f"\n{media_section}"
for target_dir in (output_dir, rag_source_dir):
target_dir.mkdir(parents=True, exist_ok=True)
(target_dir / file_name).write_text(content, encoding="utf-8")
return {
"source_id": record.get("source_id"),
"source_type": record.get("source_type"),
"access": "licensed_private",
"title": title,
"url": record.get("url"),
"platform_family": record.get("platform_family"),
"platform_version": record.get("platform_version"),
"platform_doc_id": record.get("platform_doc_id"),
"doc_book": record.get("doc_book"),
"doc_bookmark": record.get("doc_bookmark"),
"doc_content_id": record.get("doc_content_id"),
"doc_coordinate": record.get("doc_coordinate"),
"access_blocked": record.get("access_blocked"),
"access_findings": record.get("access_findings") or [],
"raw_file": record.get("file"),
"normalized_file": file_name,
"raw_sha256": source_hash,
"text_sha256": hashlib.sha256(text.encode("utf-8")).hexdigest(),
"chars": len(text),
"quality": quality,
"media": media,
}
def discover_orphan_src_records(records: list[dict[str, Any]], *, raw_dir: Path) -> list[dict[str, Any]]:
known_urls = {str(record.get("url") or "") for record in records}
discovered_by_url: dict[str, dict[str, Any]] = {}
for record in records:
raw_path = raw_dir / str(record.get("file") or "")
parent_url = str(record.get("url") or "")
if not raw_path.exists() or not parent_url:
continue
try:
raw_bytes = raw_path.read_bytes()
raw_text = decode_html(raw_bytes, record)
except OSError:
continue
extractor = LinkExtractor()
extractor.feed(raw_text)
for href in extractor.links:
src_url = normalize_url(parent_url, href)
if not src_url or not is_content_src_url(src_url) or src_url.endswith("#_print"):
continue
if src_url in known_urls or src_url in discovered_by_url:
continue
src_file = raw_slug_for_url(src_url)
src_path = raw_dir / src_file
if not src_path.exists():
continue
discovered = dict(record)
discovered.update(
{
"url": src_url,
"title": record.get("title") or record.get("source_title") or src_url,
"depth": int(record.get("depth") or 0) + 1,
"file": src_file,
"bytes": src_path.stat().st_size,
"sha256": hashlib.sha256(src_path.read_bytes()).hexdigest(),
"discovered_from_parent_url": parent_url,
"discovered_from_parent_file": record.get("file"),
}
)
discovered_by_url[src_url] = discovered
return sorted(discovered_by_url.values(), key=lambda item: str(item.get("url") or ""))
def normalize_manifest(manifest: dict[str, Any], *, raw_dir: Path, output_dir: Path, rag_source_dir: Path) -> dict[str, Any]:
pages = []
skipped = []
missing = 0
seen_text_keys: dict[tuple[str, str, str], dict[str, Any]] = {}
source_records = list(manifest.get("pages") or [])
discovered_records = discover_orphan_src_records(source_records, raw_dir=raw_dir)
all_records = [*source_records, *discovered_records]
for record in all_records:
normalized = normalize_page(record, raw_dir=raw_dir, output_dir=output_dir, rag_source_dir=rag_source_dir)
if normalized and normalized.get("skipped"):
skipped.append(normalized)
elif normalized:
text_key = (
str(normalized.get("text_sha256") or ""),
str(normalized.get("platform_version") or ""),
str(normalized.get("title") or ""),
)
if text_key[0] and text_key in seen_text_keys:
normalized["skipped"] = True
normalized["skip_reason"] = "duplicate_normalized_text"
normalized["duplicate_of"] = seen_text_keys[text_key].get("normalized_file")
duplicate_file = str(normalized.get("normalized_file") or "")
if duplicate_file:
for directory in (output_dir, rag_source_dir):
duplicate_path = directory / duplicate_file
if duplicate_path.exists():
duplicate_path.unlink()
skipped.append(normalized)
continue
seen_text_keys[text_key] = normalized
pages.append(normalized)
else:
missing += 1
return {
"schema": "onec_its_normalized_docs_manifest.v1",
"access": "licensed_private",
"raw_manifest": manifest.get("output_dir"),
"output_dir": str(output_dir),
"rag_source_dir": str(rag_source_dir),
"page_count": len(pages),
"skipped_count": len(skipped),
"missing_count": missing,
"source_record_count": len(source_records),
"discovered_src_record_count": len(discovered_records),
"pages": pages,
"skipped_pages": skipped,
}
def clean_generated_markdown(*dirs: Path) -> None:
for directory in dirs:
if not directory.exists():
continue
for path in directory.glob("*.md"):
try:
text = path.read_text(encoding="utf-8", errors="ignore")
except OSError:
continue
if 'source: "official_1c_its"' in text[:600]:
path.unlink()
def main() -> int:
parser = argparse.ArgumentParser(description="Normalize private 1C:ITS raw HTML pages into Markdown RAG sources.")
parser.add_argument("--manifest", type=Path, default=DEFAULT_RAW_MANIFEST)
parser.add_argument("--raw-dir", type=Path, default=DEFAULT_RAW_DIR)
parser.add_argument("--output-dir", type=Path, default=DEFAULT_OUTPUT_DIR)
parser.add_argument("--rag-source-dir", type=Path, default=DEFAULT_RAG_SOURCE_DIR)
parser.add_argument("--manifest-output", type=Path, default=DEFAULT_MANIFEST_OUTPUT)
parser.add_argument("--no-clean", action="store_true", help="Do not remove previously generated official 1C:ITS Markdown files.")
args = parser.parse_args()
raw_manifest = load_json(args.manifest)
if not args.no_clean:
clean_generated_markdown(args.output_dir, args.rag_source_dir)
result = normalize_manifest(raw_manifest, raw_dir=args.raw_dir, output_dir=args.output_dir, rag_source_dir=args.rag_source_dir)
args.manifest_output.parent.mkdir(parents=True, exist_ok=True)
args.manifest_output.write_text(json.dumps(result, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
print(json.dumps({"output": str(args.manifest_output), "pages": result["page_count"], "skipped": result["skipped_count"], "missing": result["missing_count"]}, ensure_ascii=False))
return 0 if result["page_count"] or result["skipped_count"] else 2
if __name__ == "__main__":
raise SystemExit(main())