389 lines
14 KiB
Python
389 lines
14 KiB
Python
from __future__ import annotations
|
||
|
||
import argparse
|
||
import hashlib
|
||
import json
|
||
import os
|
||
import re
|
||
import time
|
||
import urllib.parse
|
||
import urllib.error
|
||
from collections import deque
|
||
from html.parser import HTMLParser
|
||
from pathlib import Path
|
||
from typing import Any
|
||
|
||
from fetch_1c_its_docs import (
|
||
LinkParser,
|
||
charset_from_content_type,
|
||
fetch_url,
|
||
is_content_src_url,
|
||
normalize_url,
|
||
page_record,
|
||
url_priority,
|
||
)
|
||
from one_c_its_platform import materialize_doc_url, merge_platform_metadata, parse_doc_coordinate
|
||
|
||
|
||
ROOT = Path(__file__).resolve().parents[1]
|
||
DEFAULT_RAW_DIR = ROOT / "plugins" / "1c" / "rag" / "official-docs" / "raw"
|
||
DEFAULT_MANIFEST = DEFAULT_RAW_DIR / "manifest.json"
|
||
|
||
|
||
BOOK_SOURCE_TYPES = {
|
||
"dev": "official_1c_its_developer_guide",
|
||
"adm": "official_1c_its_admin_guide",
|
||
"cs": "official_1c_its_client_server_admin_guide",
|
||
"usr": "official_1c_its_user_guide",
|
||
"utx": "official_1c_its_taxi_user_guide",
|
||
}
|
||
|
||
|
||
class AccessMarkerParser(HTMLParser):
|
||
def __init__(self) -> None:
|
||
super().__init__(convert_charrefs=True)
|
||
self.login_links = 0
|
||
self.paywall_markers = 0
|
||
self.data_access_false = 0
|
||
self.iframe_srcs: list[str] = []
|
||
|
||
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
||
values = {name.lower(): value or "" for name, value in attrs}
|
||
href = values.get("href", "")
|
||
class_name = values.get("class", "")
|
||
if "/user/auth" in href:
|
||
self.login_links += 1
|
||
if "paywall" in class_name:
|
||
self.paywall_markers += 1
|
||
if values.get("data-access") == "false":
|
||
self.data_access_false += 1
|
||
if tag.lower() == "iframe" and values.get("src"):
|
||
self.iframe_srcs.append(values["src"])
|
||
|
||
|
||
def load_json(path: Path) -> dict[str, Any]:
|
||
if not path.exists():
|
||
return {}
|
||
data = json.loads(path.read_text(encoding="utf-8-sig"))
|
||
return data if isinstance(data, dict) else {}
|
||
|
||
|
||
def write_json(path: Path, data: dict[str, Any]) -> None:
|
||
path.parent.mkdir(parents=True, exist_ok=True)
|
||
path.write_text(json.dumps(data, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
||
|
||
|
||
def read_cookie(cookie_file: Path | None) -> str:
|
||
if cookie_file:
|
||
return cookie_file.read_text(encoding="utf-8").strip()
|
||
return os.environ.get("ONEC_ITS_COOKIE", "").strip()
|
||
|
||
|
||
def access_markers_from_html(url: str, body: bytes, content_type: str) -> dict[str, Any]:
|
||
text = body.decode(charset_from_content_type(content_type), errors="replace")
|
||
parser = AccessMarkerParser()
|
||
parser.feed(text)
|
||
iframe_src = ""
|
||
for src in parser.iframe_srcs:
|
||
if "/db/content/" in src and "/src/" in src:
|
||
iframe_src = urllib.parse.urljoin(url, src)
|
||
break
|
||
findings = []
|
||
if parser.login_links:
|
||
findings.append("login_links_present")
|
||
if parser.paywall_markers:
|
||
findings.append("paywall_marker_present")
|
||
if parser.data_access_false:
|
||
findings.append("data_access_false")
|
||
return {
|
||
"access_blocked": bool(findings),
|
||
"access_findings": findings,
|
||
"login_links": parser.login_links,
|
||
"paywall_markers": parser.paywall_markers,
|
||
"data_access_false": parser.data_access_false,
|
||
"iframe_src": iframe_src,
|
||
}
|
||
|
||
|
||
def dedupe_errors(errors: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
||
seen = set()
|
||
result = []
|
||
for error in errors:
|
||
key = (str(error.get("source_id") or ""), str(error.get("url") or ""), str(error.get("error") or ""))
|
||
if key in seen:
|
||
continue
|
||
seen.add(key)
|
||
result.append(error)
|
||
return result
|
||
|
||
|
||
def source_from_url(url: str, title: str | None = None) -> dict[str, Any]:
|
||
coord = parse_doc_coordinate(url)
|
||
doc_id = coord.get("platform_doc_id") or "its_doc"
|
||
book = coord.get("doc_book") or ""
|
||
source_type = BOOK_SOURCE_TYPES.get(book, "official_1c_its_platform_doc")
|
||
platform_version = coord.get("platform_version")
|
||
source_id_parts = [doc_id]
|
||
if book:
|
||
source_id_parts.append(book)
|
||
if coord.get("doc_bookmark") or coord.get("doc_content_id"):
|
||
source_id_parts.append(str(coord.get("doc_bookmark") or coord.get("doc_content_id")))
|
||
source_id = "_target_" + "_".join(re.sub(r"[^A-Za-zА-Яа-яЁё0-9_.-]+", "_", part) for part in source_id_parts)
|
||
source = {
|
||
"id": source_id,
|
||
"title": title or f"1C:ITS target {coord.get('doc_coordinate') or url}",
|
||
"source_type": source_type,
|
||
"platform_family": "1C:Enterprise 8" if doc_id else "",
|
||
"platform_version": platform_version,
|
||
"url": url,
|
||
}
|
||
return merge_platform_metadata(source, url)
|
||
|
||
|
||
def same_doc_space(url: str, seed_url: str) -> bool:
|
||
parsed = urllib.parse.urlparse(url)
|
||
seed = urllib.parse.urlparse(seed_url)
|
||
if parsed.netloc.lower() != seed.netloc.lower():
|
||
return False
|
||
seed_doc_id = parse_doc_coordinate(seed_url).get("platform_doc_id")
|
||
if seed_doc_id and parsed.path.startswith(f"/db/content/{seed_doc_id}/src/"):
|
||
return True
|
||
seed_root = "/".join(seed.path.strip("/").split("/")[:2])
|
||
path_root = "/".join(parsed.path.strip("/").split("/")[:2])
|
||
return bool(seed_root) and path_root == seed_root
|
||
|
||
|
||
def should_follow(url: str, seed_url: str) -> bool:
|
||
if not same_doc_space(url, seed_url):
|
||
return False
|
||
path = urllib.parse.urlparse(url).path
|
||
return (
|
||
is_content_src_url(url)
|
||
or re.search(r"/db/[^/]+/content/\d+/hdoc(?:/\d+)?$", path)
|
||
or re.search(r"/db/[^/]+/content/\d+/\d+$", path)
|
||
or "/bookmark/" in path
|
||
)
|
||
|
||
|
||
def target_link_url(base_url: str, href: str) -> str | None:
|
||
if href.startswith("#TI"):
|
||
coord = parse_doc_coordinate(base_url)
|
||
doc_id = coord.get("platform_doc_id")
|
||
book = coord.get("doc_book")
|
||
if doc_id and book:
|
||
bookmark = href.lstrip("#")
|
||
parsed = urllib.parse.urlparse(base_url)
|
||
path = f"/db/{doc_id}/bookmark/{book}/{bookmark}"
|
||
return urllib.parse.urlunparse(parsed._replace(path=path, query="", fragment=""))
|
||
return normalize_url(base_url, href)
|
||
|
||
|
||
def hdoc_variant(url: str) -> str | None:
|
||
parsed = urllib.parse.urlparse(url)
|
||
replaced = re.sub(r"(/db/[^/]+/content/\d+)/1$", r"\1/hdoc", parsed.path)
|
||
if replaced == parsed.path:
|
||
return None
|
||
return urllib.parse.urlunparse(parsed._replace(path=replaced, query="", fragment=""))
|
||
|
||
|
||
def parse_links(record: dict[str, Any], raw_dir: Path) -> list[str]:
|
||
path = raw_dir / str(record.get("file") or "")
|
||
if not path.exists():
|
||
return []
|
||
raw = path.read_bytes()
|
||
text = raw.decode(charset_from_content_type(str(record.get("content_type") or "")), errors="replace")
|
||
parser = LinkParser()
|
||
parser.feed(text)
|
||
return parser.links
|
||
|
||
|
||
def enrich_record_access_markers(record: dict[str, Any], raw_dir: Path) -> dict[str, Any]:
|
||
if record.get("access_findings") is not None:
|
||
return record
|
||
path = raw_dir / str(record.get("file") or "")
|
||
if not path.exists():
|
||
return record
|
||
try:
|
||
markers = access_markers_from_html(str(record.get("url") or ""), path.read_bytes(), str(record.get("content_type") or ""))
|
||
except OSError:
|
||
return record
|
||
updated = dict(record)
|
||
updated.update(markers)
|
||
return updated
|
||
|
||
|
||
def fetch_target(
|
||
*,
|
||
url: str,
|
||
output_dir: Path,
|
||
manifest_path: Path,
|
||
cookie: str,
|
||
max_pages: int,
|
||
max_depth: int,
|
||
timeout: int,
|
||
user_agent: str,
|
||
) -> dict[str, Any]:
|
||
output_dir.mkdir(parents=True, exist_ok=True)
|
||
materialized_url = materialize_doc_url(url)
|
||
source = source_from_url(url)
|
||
existing_manifest = load_json(manifest_path)
|
||
pages_by_url = {
|
||
str(record.get("url")): record
|
||
for record in existing_manifest.get("pages") or []
|
||
if record.get("url")
|
||
}
|
||
errors = list(existing_manifest.get("errors") or [])
|
||
target_errors = []
|
||
fetched = []
|
||
reused = []
|
||
seen: set[str] = set()
|
||
queue: deque[tuple[str, int, str | None]] = deque([(materialized_url, 0, None)])
|
||
|
||
while queue and len(fetched) + len(reused) < max_pages:
|
||
current_url, depth, title_hint = queue.popleft()
|
||
if current_url in seen:
|
||
continue
|
||
seen.add(current_url)
|
||
record = pages_by_url.get(current_url)
|
||
if record:
|
||
record = enrich_record_access_markers(record, output_dir)
|
||
pages_by_url[current_url] = record
|
||
reused.append(current_url)
|
||
else:
|
||
try:
|
||
body, headers, status = fetch_url(current_url, cookie=cookie, timeout=timeout, user_agent=user_agent)
|
||
record = page_record(
|
||
source=source,
|
||
url=current_url,
|
||
depth=depth,
|
||
body=body,
|
||
headers=headers,
|
||
status=status,
|
||
output_dir=output_dir,
|
||
title_hint=title_hint,
|
||
)
|
||
record.update(access_markers_from_html(current_url, body, str(headers.get("Content-Type") or "")))
|
||
pages_by_url[current_url] = {key: value for key, value in record.items() if key != "links"}
|
||
fetched.append(current_url)
|
||
except (urllib.error.URLError, TimeoutError, OSError, UnicodeDecodeError) as exc:
|
||
error = {"source_id": source.get("id"), "url": current_url, "depth": depth, "error": str(exc)}
|
||
errors.append(error)
|
||
target_errors.append(error)
|
||
continue
|
||
|
||
if depth >= max_depth:
|
||
continue
|
||
links = record.get("links") if "links" in record else parse_links(record, output_dir)
|
||
candidates = []
|
||
current_hdoc = hdoc_variant(current_url)
|
||
if current_hdoc and current_hdoc not in seen and should_follow(current_hdoc, materialized_url):
|
||
candidates.append(current_hdoc)
|
||
for href in links or []:
|
||
next_url = target_link_url(current_url, href)
|
||
if not next_url or next_url in seen or not should_follow(next_url, materialized_url):
|
||
continue
|
||
candidates.append(next_url)
|
||
next_hdoc = hdoc_variant(next_url)
|
||
if next_hdoc and next_hdoc not in seen and should_follow(next_hdoc, materialized_url):
|
||
candidates.append(next_hdoc)
|
||
for next_url in sorted(set(candidates), key=url_priority):
|
||
queue.append((next_url, depth + 1, str(record.get("title") or title_hint or "")))
|
||
|
||
errors = dedupe_errors(errors)
|
||
target_errors = dedupe_errors(target_errors)
|
||
pages = sorted(pages_by_url.values(), key=lambda item: str(item.get("url") or ""))
|
||
access_blocked_pages = [
|
||
{
|
||
"url": page.get("url"),
|
||
"findings": page.get("access_findings") or [],
|
||
"iframe_src": page.get("iframe_src") or "",
|
||
}
|
||
for page in pages
|
||
if page.get("access_blocked")
|
||
]
|
||
result = {
|
||
"schema": existing_manifest.get("schema") or "onec_its_raw_fetch_manifest.v1",
|
||
"access": existing_manifest.get("access") or "licensed_private",
|
||
"output_dir": str(output_dir),
|
||
"page_count": len(pages),
|
||
"error_count": len(errors),
|
||
"resume": existing_manifest.get("resume") or {"enabled": True},
|
||
"stop": {
|
||
"status": "complete",
|
||
"reason": "target_fetch_complete",
|
||
"max_pages": max_pages,
|
||
"target_url": url,
|
||
"materialized_url": materialized_url,
|
||
"access_blocked_pages": len(access_blocked_pages),
|
||
},
|
||
"records_by_source": existing_manifest.get("records_by_source") or {},
|
||
"errors_by_source": existing_manifest.get("errors_by_source") or {},
|
||
"pages": pages,
|
||
"errors": errors,
|
||
}
|
||
write_json(manifest_path, result)
|
||
progress_path = manifest_path.parent / "progress.json"
|
||
write_json(
|
||
progress_path,
|
||
{
|
||
"schema": "onec_its_target_fetch_progress.v1",
|
||
"status": "complete",
|
||
"updated_at_unix": int(time.time()),
|
||
"target_url": url,
|
||
"materialized_url": materialized_url,
|
||
"source": source,
|
||
"fetched": fetched,
|
||
"reused": reused,
|
||
"errors": target_errors,
|
||
"access_blocked_pages": access_blocked_pages[-20:],
|
||
"page_count": len(pages),
|
||
"error_count": len(errors),
|
||
},
|
||
)
|
||
return {
|
||
"target_url": url,
|
||
"materialized_url": materialized_url,
|
||
"source": source,
|
||
"fetched_count": len(fetched),
|
||
"reused_count": len(reused),
|
||
"error_count": len(target_errors),
|
||
"access_blocked_count": len(access_blocked_pages),
|
||
"access_blocked_pages": access_blocked_pages[-20:],
|
||
"fetched": fetched,
|
||
"reused": reused,
|
||
"errors": target_errors,
|
||
"manifest": str(manifest_path),
|
||
"progress": str(progress_path),
|
||
}
|
||
|
||
|
||
def main() -> int:
|
||
parser = argparse.ArgumentParser(description="Fetch a targeted 1C:ITS documentation page and merge it into the raw manifest.")
|
||
parser.add_argument("url")
|
||
parser.add_argument("--output-dir", type=Path, default=DEFAULT_RAW_DIR)
|
||
parser.add_argument("--manifest", type=Path, default=DEFAULT_MANIFEST)
|
||
parser.add_argument("--cookie-file", type=Path)
|
||
parser.add_argument("--max-pages", type=int, default=12)
|
||
parser.add_argument("--max-depth", type=int, default=2)
|
||
parser.add_argument("--timeout", type=int, default=30)
|
||
parser.add_argument("--user-agent", default="Codex 1C ITS target fetch")
|
||
parser.add_argument("--print", action="store_true", dest="print_report")
|
||
args = parser.parse_args()
|
||
|
||
result = fetch_target(
|
||
url=args.url,
|
||
output_dir=args.output_dir,
|
||
manifest_path=args.manifest,
|
||
cookie=read_cookie(args.cookie_file),
|
||
max_pages=max(1, args.max_pages),
|
||
max_depth=max(0, args.max_depth),
|
||
timeout=args.timeout,
|
||
user_agent=args.user_agent,
|
||
)
|
||
print(json.dumps(result if args.print_report else {key: result[key] for key in ("target_url", "materialized_url", "fetched_count", "reused_count", "error_count")}, ensure_ascii=False, indent=2))
|
||
return 0 if result["error_count"] == 0 or result["fetched_count"] or result["reused_count"] else 1
|
||
|
||
|
||
if __name__ == "__main__":
|
||
raise SystemExit(main())
|