Initial SQL-only 1C adapter baseline
This commit is contained in:
@@ -0,0 +1,547 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
import urllib.error
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
from collections import deque
|
||||
from html.parser import HTMLParser
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import yaml
|
||||
|
||||
from one_c_its_platform import merge_platform_metadata
|
||||
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
DEFAULT_SOURCES = ROOT / "plugins" / "1c" / "rag" / "official-docs" / "sources.yaml"
|
||||
DEFAULT_OUTPUT_DIR = ROOT / "plugins" / "1c" / "rag" / "official-docs" / "raw"
|
||||
DEFAULT_MANIFEST = DEFAULT_OUTPUT_DIR / "manifest.json"
|
||||
DEFAULT_PROGRESS = DEFAULT_OUTPUT_DIR / "progress.json"
|
||||
|
||||
|
||||
class LinkParser(HTMLParser):
|
||||
def __init__(self) -> None:
|
||||
super().__init__()
|
||||
self.links: list[str] = []
|
||||
self.title_parts: list[str] = []
|
||||
self._in_title = False
|
||||
|
||||
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
||||
tag = tag.lower()
|
||||
if tag == "title":
|
||||
self._in_title = True
|
||||
attr_name = "href" if tag == "a" else "src" if tag in {"iframe", "frame"} else None
|
||||
if attr_name is None:
|
||||
return
|
||||
for name, value in attrs:
|
||||
if name.lower() == attr_name and value:
|
||||
self.links.append(value)
|
||||
|
||||
def handle_endtag(self, tag: str) -> None:
|
||||
if tag.lower() == "title":
|
||||
self._in_title = False
|
||||
|
||||
def handle_data(self, data: str) -> None:
|
||||
if self._in_title:
|
||||
self.title_parts.append(data.strip())
|
||||
|
||||
@property
|
||||
def title(self) -> str:
|
||||
return " ".join(part for part in self.title_parts if part).strip()
|
||||
|
||||
|
||||
def charset_from_content_type(content_type: str) -> str:
|
||||
match = re.search(r"charset=([^;\s]+)", content_type, flags=re.IGNORECASE)
|
||||
if match:
|
||||
return match.group(1).strip("\"'")
|
||||
return "utf-8"
|
||||
|
||||
|
||||
def load_yaml(path: Path) -> dict[str, Any]:
|
||||
data = yaml.safe_load(path.read_text(encoding="utf-8"))
|
||||
if not isinstance(data, dict):
|
||||
raise ValueError(f"{path} must contain a YAML mapping")
|
||||
return data
|
||||
|
||||
|
||||
def read_cookie(cookie_file: Path | None) -> str:
|
||||
if cookie_file:
|
||||
return cookie_file.read_text(encoding="utf-8").strip()
|
||||
return os.environ.get("ONEC_ITS_COOKIE", "").strip()
|
||||
|
||||
|
||||
def safe_slug(value: str) -> str:
|
||||
parsed = urllib.parse.urlparse(value)
|
||||
raw = f"{parsed.netloc}{parsed.path}".strip("/") or "index"
|
||||
if parsed.query:
|
||||
raw += "_" + parsed.query
|
||||
slug = re.sub(r"[^A-Za-zА-Яа-яЁё0-9_.-]+", "_", raw, flags=re.UNICODE).strip("_")
|
||||
digest = hashlib.sha256(value.encode("utf-8")).hexdigest()[:12]
|
||||
return f"{slug[:120]}__{digest}.html"
|
||||
|
||||
|
||||
def normalize_url(base_url: str, href: str) -> str | None:
|
||||
if href.startswith(("mailto:", "tel:", "javascript:")):
|
||||
return None
|
||||
absolute = urllib.parse.urljoin(base_url, href)
|
||||
parsed = urllib.parse.urlparse(absolute)
|
||||
if parsed.scheme not in {"http", "https"}:
|
||||
return None
|
||||
parsed = parsed._replace(fragment="")
|
||||
return urllib.parse.urlunparse(parsed)
|
||||
|
||||
|
||||
def request_safe_url(url: str) -> str:
|
||||
parsed = urllib.parse.urlparse(url)
|
||||
path = urllib.parse.quote(urllib.parse.unquote(parsed.path), safe="/")
|
||||
query = urllib.parse.quote(urllib.parse.unquote(parsed.query), safe="=&?/:;%[]@!$'()*+,")
|
||||
return urllib.parse.urlunparse(parsed._replace(path=path, query=query))
|
||||
|
||||
|
||||
def compile_patterns(patterns: list[str] | None) -> list[re.Pattern[str]]:
|
||||
return [re.compile(pattern) for pattern in patterns or []]
|
||||
|
||||
|
||||
def allowed_url(url: str, *, seed_host: str, policy: dict[str, Any]) -> bool:
|
||||
parsed = urllib.parse.urlparse(url)
|
||||
if policy.get("same_host", True) and parsed.netloc.lower() != seed_host.lower():
|
||||
return False
|
||||
include = compile_patterns(policy.get("include_patterns"))
|
||||
exclude = compile_patterns(policy.get("exclude_patterns"))
|
||||
if include and not any(pattern.search(url) for pattern in include):
|
||||
return False
|
||||
if any(pattern.search(url) for pattern in exclude):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def url_priority(url: str) -> int:
|
||||
parsed = urllib.parse.urlparse(url)
|
||||
path = parsed.path
|
||||
if is_content_src_url(url):
|
||||
return 0
|
||||
if re.search(r"/db/[^/]+/content/\d+/hdoc(?:/\d+)?$", path):
|
||||
return 1
|
||||
if re.search(r"/db/[^/]+/content/\d+/\d+$", path):
|
||||
return 1
|
||||
return 2
|
||||
|
||||
|
||||
def is_content_src_url(url: str) -> bool:
|
||||
path = urllib.parse.urlparse(url).path
|
||||
return "/db/content/" in path and "/src/" in path
|
||||
|
||||
|
||||
def enqueue_links(
|
||||
queue: deque[tuple[dict[str, Any], str, int, dict[str, Any], str | None]],
|
||||
*,
|
||||
source: dict[str, Any],
|
||||
base_url: str,
|
||||
depth: int,
|
||||
policy: dict[str, Any],
|
||||
links: list[str],
|
||||
seen: set[str],
|
||||
title_hint: str | None = None,
|
||||
src_only: bool = False,
|
||||
front_priority: bool = False,
|
||||
) -> int:
|
||||
seed_host = urllib.parse.urlparse(source["url"]).netloc
|
||||
candidates = []
|
||||
queued_urls = {item[1] for item in queue}
|
||||
for href in links:
|
||||
next_url = normalize_url(base_url, href)
|
||||
if not next_url or next_url in seen or next_url in queued_urls:
|
||||
continue
|
||||
if src_only and not is_content_src_url(next_url):
|
||||
continue
|
||||
if not allowed_url(next_url, seed_host=seed_host, policy=policy):
|
||||
continue
|
||||
candidates.append(next_url)
|
||||
queued_urls.add(next_url)
|
||||
candidates.sort(key=url_priority)
|
||||
|
||||
# Real article bodies live under /db/content/.../src/... and hdoc pages
|
||||
# usually contain the iframe pointing there, so put those before nav links.
|
||||
priority_items = [url for url in candidates if url_priority(url) <= 1]
|
||||
normal_items = [url for url in candidates if url_priority(url) > 1]
|
||||
if front_priority:
|
||||
for next_url in reversed(priority_items):
|
||||
hint = title_hint if is_content_src_url(next_url) else None
|
||||
queue.appendleft((source, next_url, depth + 1, policy, hint))
|
||||
else:
|
||||
for next_url in priority_items:
|
||||
hint = title_hint if is_content_src_url(next_url) else None
|
||||
queue.append((source, next_url, depth + 1, policy, hint))
|
||||
for next_url in normal_items:
|
||||
queue.append((source, next_url, depth + 1, policy, None))
|
||||
return len(candidates)
|
||||
|
||||
|
||||
def merged_policy(default_policy: dict[str, Any], source_policy: dict[str, Any] | None) -> dict[str, Any]:
|
||||
policy = dict(default_policy)
|
||||
for key, value in (source_policy or {}).items():
|
||||
if key in {"include_patterns", "exclude_patterns"}:
|
||||
policy[key] = [*policy.get(key, []), *value]
|
||||
else:
|
||||
policy[key] = value
|
||||
return policy
|
||||
|
||||
|
||||
def fetch_url(url: str, *, cookie: str, timeout: int, user_agent: str) -> tuple[bytes, dict[str, str], int]:
|
||||
headers = {"User-Agent": user_agent}
|
||||
if cookie:
|
||||
headers["Cookie"] = cookie
|
||||
request = urllib.request.Request(request_safe_url(url), headers=headers)
|
||||
with urllib.request.urlopen(request, timeout=timeout) as response:
|
||||
body = response.read()
|
||||
response_headers = {key: value for key, value in response.headers.items()}
|
||||
return body, response_headers, int(response.status)
|
||||
|
||||
|
||||
def load_json(path: Path) -> dict[str, Any]:
|
||||
if not path.exists():
|
||||
return {}
|
||||
try:
|
||||
data = json.loads(path.read_text(encoding="utf-8-sig"))
|
||||
except json.JSONDecodeError:
|
||||
return {}
|
||||
return data if isinstance(data, dict) else {}
|
||||
|
||||
|
||||
def resumable_pages(*, output_dir: Path, manifest_path: Path, progress_path: Path) -> dict[str, dict[str, Any]]:
|
||||
pages_by_url: dict[str, dict[str, Any]] = {}
|
||||
for source in (load_json(manifest_path), load_json(progress_path)):
|
||||
for record in source.get("pages") or []:
|
||||
url = record.get("url")
|
||||
filename = record.get("file")
|
||||
sha256 = record.get("sha256")
|
||||
if not url or not filename or not sha256:
|
||||
continue
|
||||
path = output_dir / str(filename)
|
||||
if not path.exists():
|
||||
continue
|
||||
if hashlib.sha256(path.read_bytes()).hexdigest() != sha256:
|
||||
continue
|
||||
pages_by_url[str(url)] = record
|
||||
return pages_by_url
|
||||
|
||||
|
||||
def write_progress(
|
||||
path: Path,
|
||||
*,
|
||||
config: dict[str, Any],
|
||||
output_dir: Path,
|
||||
max_pages: int,
|
||||
records: list[dict[str, Any]],
|
||||
errors: list[dict[str, Any]],
|
||||
seen: set[str],
|
||||
queue_size: int,
|
||||
current_url: str | None,
|
||||
resumed_pages: int,
|
||||
skipped_existing: int,
|
||||
status: str,
|
||||
stop_reason: str | None = None,
|
||||
) -> None:
|
||||
progress = {
|
||||
"schema": "onec_its_fetch_progress.v1",
|
||||
"status": status,
|
||||
"updated_at_unix": int(time.time()),
|
||||
"access": config.get("access") or "licensed_private",
|
||||
"output_dir": str(output_dir),
|
||||
"max_pages": max_pages,
|
||||
"page_count": len(records),
|
||||
"error_count": len(errors),
|
||||
"seen_count": len(seen),
|
||||
"queue_size": queue_size,
|
||||
"current_url": current_url,
|
||||
"stop_reason": stop_reason,
|
||||
"resumed_pages": resumed_pages,
|
||||
"skipped_existing": skipped_existing,
|
||||
"pages": records,
|
||||
"errors": errors,
|
||||
}
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(json.dumps(progress, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
||||
|
||||
|
||||
def page_record(
|
||||
*,
|
||||
source: dict[str, Any],
|
||||
url: str,
|
||||
depth: int,
|
||||
body: bytes,
|
||||
headers: dict[str, str],
|
||||
status: int,
|
||||
output_dir: Path,
|
||||
title_hint: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
parser = LinkParser()
|
||||
text = body.decode(charset_from_content_type(headers.get("Content-Type", "")), errors="replace")
|
||||
parser.feed(text)
|
||||
filename = safe_slug(url)
|
||||
(output_dir / filename).write_bytes(body)
|
||||
return merge_platform_metadata({
|
||||
"source_id": source.get("id"),
|
||||
"source_title": source.get("title"),
|
||||
"source_type": source.get("source_type"),
|
||||
"platform_family": source.get("platform_family"),
|
||||
"platform_version": source.get("platform_version"),
|
||||
"access": "licensed_private",
|
||||
"url": url,
|
||||
"title": parser.title or title_hint or source.get("title") or url,
|
||||
"depth": depth,
|
||||
"status": status,
|
||||
"content_type": headers.get("Content-Type", ""),
|
||||
"file": filename,
|
||||
"bytes": len(body),
|
||||
"sha256": hashlib.sha256(body).hexdigest(),
|
||||
"fetched_at_unix": int(time.time()),
|
||||
"links": parser.links,
|
||||
}, url)
|
||||
|
||||
|
||||
def fetch_sources(
|
||||
config: dict[str, Any],
|
||||
*,
|
||||
output_dir: Path,
|
||||
cookie: str,
|
||||
max_pages: int,
|
||||
delay_seconds: float,
|
||||
timeout: int,
|
||||
user_agent: str,
|
||||
manifest_path: Path,
|
||||
progress_path: Path,
|
||||
resume: bool,
|
||||
max_attempts: int,
|
||||
) -> dict[str, Any]:
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
default_policy = config.get("default_policy") or {}
|
||||
records: list[dict[str, Any]] = []
|
||||
errors: list[dict[str, Any]] = []
|
||||
seen: set[str] = set()
|
||||
skipped_existing = 0
|
||||
resume_records = resumable_pages(output_dir=output_dir, manifest_path=manifest_path, progress_path=progress_path) if resume else {}
|
||||
records_by_url: dict[str, dict[str, Any]] = {}
|
||||
records_by_source: dict[str, int] = {}
|
||||
errors_by_source: dict[str, int] = {}
|
||||
|
||||
queue: deque[tuple[dict[str, Any], str, int, dict[str, Any], str | None]] = deque()
|
||||
seed_urls = set()
|
||||
priority_repacked = False
|
||||
for source in config.get("sources") or []:
|
||||
policy = merged_policy(default_policy, source.get("policy"))
|
||||
seed_urls.add(str(source["url"]))
|
||||
queue.append((source, source["url"], 0, policy, None))
|
||||
|
||||
while queue and len(records) < max_pages and (max_attempts <= 0 or len(seen) < max_attempts):
|
||||
if not priority_repacked and seed_urls and seed_urls.issubset(seen):
|
||||
queue = deque(sorted(queue, key=lambda item: (url_priority(item[1]), item[2], item[1])))
|
||||
priority_repacked = True
|
||||
source, url, depth, policy, title_hint = queue.popleft()
|
||||
source_id = str(source.get("id") or "")
|
||||
max_pages_per_source = int(policy.get("max_pages_per_source") or 0)
|
||||
max_errors_per_source = int(policy.get("max_errors_per_source") or 0)
|
||||
if max_pages_per_source > 0 and records_by_source.get(source_id, 0) >= max_pages_per_source:
|
||||
continue
|
||||
if max_errors_per_source > 0 and errors_by_source.get(source_id, 0) >= max_errors_per_source:
|
||||
continue
|
||||
if url in seen:
|
||||
continue
|
||||
seen.add(url)
|
||||
if resume and url in resume_records:
|
||||
record = resume_records[url]
|
||||
records_by_url[url] = record
|
||||
records = list(records_by_url.values())
|
||||
records_by_source[source_id] = records_by_source.get(source_id, 0) + 1
|
||||
skipped_existing += 1
|
||||
write_progress(
|
||||
progress_path,
|
||||
config=config,
|
||||
output_dir=output_dir,
|
||||
max_pages=max_pages,
|
||||
records=records,
|
||||
errors=errors,
|
||||
seen=seen,
|
||||
queue_size=len(queue),
|
||||
current_url=url,
|
||||
resumed_pages=len(resume_records),
|
||||
skipped_existing=skipped_existing,
|
||||
status="running",
|
||||
)
|
||||
max_depth = int(policy.get("max_depth", 0))
|
||||
try:
|
||||
body = (output_dir / str(record["file"])).read_bytes()
|
||||
text = body.decode(charset_from_content_type(str(record.get("content_type") or "")), errors="replace")
|
||||
parser = LinkParser()
|
||||
parser.feed(text)
|
||||
if depth < max_depth or any(is_content_src_url(normalize_url(url, href) or "") for href in parser.links):
|
||||
enqueue_links(
|
||||
queue,
|
||||
source=source,
|
||||
base_url=url,
|
||||
depth=depth,
|
||||
policy=policy,
|
||||
links=parser.links,
|
||||
seen=seen,
|
||||
title_hint=str(record.get("title") or title_hint or ""),
|
||||
src_only=depth >= max_depth,
|
||||
front_priority=seed_urls.issubset(seen),
|
||||
)
|
||||
except OSError as exc:
|
||||
errors.append({"source_id": source.get("id"), "url": url, "depth": depth, "error": f"resume link parse failed: {exc}"})
|
||||
continue
|
||||
try:
|
||||
write_progress(
|
||||
progress_path,
|
||||
config=config,
|
||||
output_dir=output_dir,
|
||||
max_pages=max_pages,
|
||||
records=records,
|
||||
errors=errors,
|
||||
seen=seen,
|
||||
queue_size=len(queue),
|
||||
current_url=url,
|
||||
resumed_pages=len(resume_records),
|
||||
skipped_existing=skipped_existing,
|
||||
status="running",
|
||||
)
|
||||
body, headers, status = fetch_url(url, cookie=cookie, timeout=timeout, user_agent=user_agent)
|
||||
record = page_record(source=source, url=url, depth=depth, body=body, headers=headers, status=status, output_dir=output_dir, title_hint=title_hint)
|
||||
records_by_url[url] = {key: value for key, value in record.items() if key != "links"}
|
||||
records = list(records_by_url.values())
|
||||
records_by_source[source_id] = records_by_source.get(source_id, 0) + 1
|
||||
max_depth = int(policy.get("max_depth", 0))
|
||||
if depth < max_depth or any(is_content_src_url(normalize_url(url, href) or "") for href in record["links"]):
|
||||
enqueue_links(
|
||||
queue,
|
||||
source=source,
|
||||
base_url=url,
|
||||
depth=depth,
|
||||
policy=policy,
|
||||
links=record["links"],
|
||||
seen=seen,
|
||||
title_hint=str(record.get("title") or title_hint or ""),
|
||||
src_only=depth >= max_depth,
|
||||
front_priority=seed_urls.issubset(seen),
|
||||
)
|
||||
if delay_seconds > 0:
|
||||
time.sleep(delay_seconds)
|
||||
except (urllib.error.URLError, TimeoutError, OSError, UnicodeDecodeError) as exc:
|
||||
errors.append({"source_id": source.get("id"), "url": url, "depth": depth, "error": str(exc)})
|
||||
errors_by_source[source_id] = errors_by_source.get(source_id, 0) + 1
|
||||
write_progress(
|
||||
progress_path,
|
||||
config=config,
|
||||
output_dir=output_dir,
|
||||
max_pages=max_pages,
|
||||
records=records,
|
||||
errors=errors,
|
||||
seen=seen,
|
||||
queue_size=len(queue),
|
||||
current_url=url,
|
||||
resumed_pages=len(resume_records),
|
||||
skipped_existing=skipped_existing,
|
||||
status="running",
|
||||
)
|
||||
|
||||
if len(records) >= max_pages:
|
||||
final_status = "complete"
|
||||
stop_reason = "max_pages_reached"
|
||||
elif max_attempts > 0 and len(seen) >= max_attempts:
|
||||
final_status = "stopped_by_limit"
|
||||
stop_reason = "max_attempts_reached"
|
||||
elif not queue:
|
||||
final_status = "complete"
|
||||
stop_reason = "queue_empty"
|
||||
else:
|
||||
final_status = "empty"
|
||||
stop_reason = "no_pages"
|
||||
|
||||
result = {
|
||||
"schema": "onec_its_raw_fetch_manifest.v1",
|
||||
"access": config.get("access") or "licensed_private",
|
||||
"output_dir": str(output_dir),
|
||||
"page_count": len(records),
|
||||
"error_count": len(errors),
|
||||
"resume": {
|
||||
"enabled": resume,
|
||||
"resumable_pages": len(resume_records),
|
||||
"skipped_existing": skipped_existing,
|
||||
},
|
||||
"stop": {
|
||||
"status": final_status,
|
||||
"reason": stop_reason,
|
||||
"seen_count": len(seen),
|
||||
"queue_size": len(queue),
|
||||
"max_attempts": max_attempts,
|
||||
"max_pages": max_pages,
|
||||
},
|
||||
"records_by_source": dict(sorted(records_by_source.items())),
|
||||
"errors_by_source": dict(sorted(errors_by_source.items())),
|
||||
"pages": records,
|
||||
"errors": errors,
|
||||
}
|
||||
write_progress(
|
||||
progress_path,
|
||||
config=config,
|
||||
output_dir=output_dir,
|
||||
max_pages=max_pages,
|
||||
records=records,
|
||||
errors=errors,
|
||||
seen=seen,
|
||||
queue_size=len(queue),
|
||||
current_url=None,
|
||||
resumed_pages=len(resume_records),
|
||||
skipped_existing=skipped_existing,
|
||||
status=final_status,
|
||||
stop_reason=stop_reason,
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description="Fetch official 1C:ITS documentation pages into a private raw cache.")
|
||||
parser.add_argument("--sources", type=Path, default=DEFAULT_SOURCES)
|
||||
parser.add_argument("--output-dir", type=Path, default=DEFAULT_OUTPUT_DIR)
|
||||
parser.add_argument("--manifest", type=Path, default=DEFAULT_MANIFEST)
|
||||
parser.add_argument("--progress", type=Path, default=DEFAULT_PROGRESS)
|
||||
parser.add_argument("--cookie-file", type=Path)
|
||||
parser.add_argument("--max-pages", type=int, default=50)
|
||||
parser.add_argument("--max-attempts", type=int, default=0, help="Maximum URL attempts, including failed pages. 0 means max(max_pages * 5, max_pages).")
|
||||
parser.add_argument("--no-resume", action="store_true", help="Do not reuse already downloaded pages from manifest/progress.")
|
||||
parser.add_argument("--delay-seconds", type=float, default=0.4)
|
||||
parser.add_argument("--timeout", type=int, default=30)
|
||||
parser.add_argument("--user-agent", default="Codex-1C-RAG/1.0 (+licensed private 1C:ITS access)")
|
||||
args = parser.parse_args()
|
||||
|
||||
config = load_yaml(args.sources)
|
||||
cookie = read_cookie(args.cookie_file)
|
||||
max_attempts = args.max_attempts if args.max_attempts > 0 else max(args.max_pages * 5, args.max_pages)
|
||||
result = fetch_sources(
|
||||
config,
|
||||
output_dir=args.output_dir,
|
||||
cookie=cookie,
|
||||
max_pages=args.max_pages,
|
||||
delay_seconds=args.delay_seconds,
|
||||
timeout=args.timeout,
|
||||
user_agent=args.user_agent,
|
||||
manifest_path=args.manifest,
|
||||
progress_path=args.progress,
|
||||
resume=not args.no_resume,
|
||||
max_attempts=max_attempts,
|
||||
)
|
||||
args.manifest.parent.mkdir(parents=True, exist_ok=True)
|
||||
args.manifest.write_text(json.dumps(result, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
||||
print(json.dumps({"output": str(args.manifest), "progress": str(args.progress), "pages": result["page_count"], "errors": result["error_count"], "resume": result["resume"]}, ensure_ascii=False))
|
||||
return 0 if result["page_count"] else 2
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user