374 lines
13 KiB
Python
374 lines
13 KiB
Python
#!/usr/bin/env python3
|
|
"""Download and archive documents from the Kyrgyz Republic Ministry of Justice."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import base64
|
|
import hashlib
|
|
import json
|
|
import logging
|
|
import os
|
|
import sqlite3
|
|
import sys
|
|
import tempfile
|
|
import time
|
|
import urllib.error
|
|
import urllib.parse
|
|
import urllib.request
|
|
from dataclasses import dataclass
|
|
from datetime import datetime, timezone
|
|
from pathlib import Path
|
|
from typing import Callable, Iterable
|
|
|
|
APP_VERSION = "0.8.1"
|
|
API_BASE_URL = "https://cbd.minjust.gov.kg/api/v1/OpenData/"
|
|
LANGUAGES = {"Rus": "ru", "Kyr": "ky"}
|
|
IMAGE_LANGUAGES = {"Russian": "ru", "Kyrgyz": "ky"}
|
|
LOGGER = logging.getLogger(__name__)
|
|
|
|
|
|
class CbdClient:
|
|
def __init__(
|
|
self,
|
|
requests_per_second: float = 1,
|
|
timeout: int = 60,
|
|
retries: int = 5,
|
|
) -> None:
|
|
if requests_per_second <= 0:
|
|
raise ValueError("requests_per_second must be greater than zero")
|
|
if timeout <= 0 or retries <= 0:
|
|
raise ValueError("timeout and retries must be greater than zero")
|
|
# ponytail: per-process limit; add a distributed lock before multiple workers.
|
|
self.interval = 1 / requests_per_second
|
|
self.timeout = timeout
|
|
self.retries = retries
|
|
self.last_request = 0.0
|
|
self.total_documents = 0
|
|
|
|
def request_json(self, method: str, parameters: Iterable[tuple[str, object]] = ()):
|
|
query = urllib.parse.urlencode(list(parameters), doseq=True)
|
|
extension = "" if method == "CheckAvailable" else ".json"
|
|
url = f"{API_BASE_URL}{method}{extension}" + (f"?{query}" if query else "")
|
|
for attempt in range(self.retries):
|
|
retry_delay = 2**attempt
|
|
wait = self.interval - (time.monotonic() - self.last_request)
|
|
if wait > 0:
|
|
time.sleep(wait)
|
|
request = urllib.request.Request(
|
|
url,
|
|
headers={
|
|
"Accept": "application/json",
|
|
"User-Agent": f"Akyldash/{APP_VERSION} (+https://cbd.minjust.gov.kg)",
|
|
},
|
|
)
|
|
try:
|
|
self.last_request = time.monotonic()
|
|
with urllib.request.urlopen(request, timeout=self.timeout) as response:
|
|
body = response.read()
|
|
return json.loads(body) if body else None
|
|
except urllib.error.HTTPError as error:
|
|
if error.code != 429 and error.code < 500:
|
|
raise
|
|
if error.code == 429:
|
|
try:
|
|
retry_delay = max(
|
|
retry_delay, float(error.headers.get("Retry-After", 0))
|
|
)
|
|
except ValueError:
|
|
pass
|
|
except (TimeoutError, urllib.error.URLError):
|
|
pass
|
|
if attempt + 1 < self.retries:
|
|
time.sleep(retry_delay)
|
|
raise RuntimeError(f"Ministry of Justice API request failed: {url}")
|
|
|
|
def check_available(self) -> None:
|
|
self.request_json("CheckAvailable")
|
|
|
|
def document_codes(self, limit: int | None = None, page_size: int = 1000):
|
|
def query_page(page_number: int):
|
|
return self.request_json(
|
|
"GetDocumentListByQuery",
|
|
(
|
|
("Property", "Code"),
|
|
("PageSize", page_size),
|
|
("PageNumber", page_number),
|
|
),
|
|
)
|
|
|
|
first = query_page(1)
|
|
yielded = 0
|
|
page = first
|
|
page_number = 1
|
|
self.total_documents = min(first["TotalCount"], limit or first["TotalCount"])
|
|
while True:
|
|
for document in page["Documents"]:
|
|
yield document["Code"]
|
|
yielded += 1
|
|
if limit is not None and yielded >= limit:
|
|
return
|
|
if yielded >= page["TotalCount"]:
|
|
return
|
|
page_number += 1
|
|
try:
|
|
page = self.request_json(
|
|
"GetDocumentListById",
|
|
(
|
|
("DocumentListId", first["Id"]),
|
|
("Property", "Code"),
|
|
("PageSize", page_size),
|
|
("PageNumber", page_number),
|
|
),
|
|
)
|
|
except (urllib.error.HTTPError, RuntimeError) as error:
|
|
if isinstance(error, urllib.error.HTTPError) and error.code != 404:
|
|
raise
|
|
first = page = query_page(page_number)
|
|
self.total_documents = min(
|
|
first["TotalCount"], limit or first["TotalCount"]
|
|
)
|
|
|
|
def document(self, code: int) -> dict:
|
|
return self.request_json(
|
|
"GetDocument",
|
|
(
|
|
("Code", code),
|
|
("Editions.Select", "all"),
|
|
("Editions.Data", "all"),
|
|
("Editions.Images.Select", "all"),
|
|
("Editions.Images.Data", "all"),
|
|
),
|
|
)
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class SyncResult:
|
|
discovered: int = 0
|
|
downloaded: int = 0
|
|
skipped: int = 0
|
|
failed: int = 0
|
|
|
|
|
|
def format_duration(seconds: float) -> str:
|
|
seconds = max(0, round(seconds))
|
|
hours, seconds = divmod(seconds, 3600)
|
|
minutes, seconds = divmod(seconds, 60)
|
|
return f"{hours:02d}:{minutes:02d}:{seconds:02d}"
|
|
|
|
|
|
def progress_line(
|
|
result: SyncResult, total: int, elapsed: float, width: int = 24
|
|
) -> str:
|
|
processed = result.discovered
|
|
fraction = processed / total if total else 0
|
|
filled = min(width, round(width * fraction))
|
|
rate = processed / elapsed if elapsed > 0 else 0
|
|
eta = (total - processed) / rate if rate else 0
|
|
return (
|
|
f"[{'#' * filled}{'-' * (width - filled)}] {fraction:6.2%} "
|
|
f"{processed}/{total} downloaded={result.downloaded} "
|
|
f"skipped={result.skipped} failed={result.failed} "
|
|
f"rate={rate:.2f}/s ETA={format_duration(eta)}"
|
|
)
|
|
|
|
|
|
def terminal_progress() -> Callable[[SyncResult, int], None]:
|
|
started = time.monotonic()
|
|
|
|
def update(result: SyncResult, total: int) -> None:
|
|
line = progress_line(result, total, time.monotonic() - started)
|
|
print(
|
|
f"\r{line}",
|
|
end="\n" if result.discovered >= total else "",
|
|
file=sys.stderr,
|
|
flush=True,
|
|
)
|
|
|
|
return update
|
|
|
|
|
|
def connect_manifest(path: Path) -> sqlite3.Connection:
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
connection = sqlite3.connect(path)
|
|
connection.execute(
|
|
"""
|
|
CREATE TABLE IF NOT EXISTS documents (
|
|
code INTEGER PRIMARY KEY,
|
|
source_sha256 TEXT NOT NULL,
|
|
fetched_at TEXT NOT NULL
|
|
)
|
|
"""
|
|
)
|
|
connection.execute(
|
|
"""
|
|
CREATE TABLE IF NOT EXISTS errors (
|
|
code INTEGER PRIMARY KEY,
|
|
message TEXT NOT NULL,
|
|
failed_at TEXT NOT NULL
|
|
)
|
|
"""
|
|
)
|
|
return connection
|
|
|
|
|
|
def atomic_write(path: Path, content: bytes) -> None:
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
with tempfile.NamedTemporaryFile(dir=path.parent, delete=False) as temporary:
|
|
temporary.write(content)
|
|
temporary_path = Path(temporary.name)
|
|
os.replace(temporary_path, path)
|
|
|
|
|
|
def json_bytes(value: object) -> bytes:
|
|
return (json.dumps(value, ensure_ascii=False, indent=2) + "\n").encode()
|
|
|
|
|
|
def archive_document(root: Path, document: dict) -> str:
|
|
code = int(document["Code"])
|
|
source = json.dumps(document, ensure_ascii=False, sort_keys=True).encode()
|
|
source_sha256 = hashlib.sha256(source).hexdigest()
|
|
directory = root / "documents" / str(code)
|
|
editions = document.get("Editions") or []
|
|
metadata = {key: value for key, value in document.items() if key != "Editions"}
|
|
atomic_write(directory / "metadata.json", json_bytes(metadata))
|
|
|
|
for edition in editions:
|
|
edition_directory = directory / "editions" / str(int(edition["Code"]))
|
|
edition_metadata = {key: value for key, value in edition.items() if key != "Data"}
|
|
edition_metadata["Images"] = [
|
|
{key: value for key, value in image.items() if key != "Data"}
|
|
for image in edition.get("Images") or []
|
|
]
|
|
atomic_write(edition_directory / "metadata.json", json_bytes(edition_metadata))
|
|
|
|
for source_language, filename in LANGUAGES.items():
|
|
html = (edition.get("Data") or {}).get(source_language)
|
|
if html:
|
|
atomic_write(edition_directory / f"{filename}.html", html.encode())
|
|
|
|
for image in edition.get("Images") or []:
|
|
if not image.get("Data"):
|
|
continue
|
|
language = IMAGE_LANGUAGES.get(image.get("Lang"), "unknown")
|
|
filename = Path(str(image.get("Name") or "image").replace("\\", "/")).name
|
|
if filename in {"", ".", ".."}:
|
|
raise ValueError(f"Invalid image filename in document {code}")
|
|
atomic_write(
|
|
edition_directory / "images" / language / filename,
|
|
base64.b64decode(image["Data"], validate=True),
|
|
)
|
|
return source_sha256
|
|
|
|
|
|
def sync_archive(
|
|
output: Path,
|
|
client: CbdClient | None = None,
|
|
limit: int | None = None,
|
|
refresh: bool = False,
|
|
progress: Callable[[SyncResult, int], None] | None = None,
|
|
) -> SyncResult:
|
|
client = client or CbdClient()
|
|
output.mkdir(parents=True, exist_ok=True)
|
|
connection = connect_manifest(output / "manifest.sqlite3")
|
|
known = {row[0] for row in connection.execute("SELECT code FROM documents")}
|
|
discovered = downloaded = skipped = failed = 0
|
|
|
|
client.check_available()
|
|
for code in client.document_codes(limit):
|
|
discovered += 1
|
|
# ponytail: refresh scans all documents; use sitemap lastmod when scheduled
|
|
# update checks become frequent enough for the extra mapping logic to pay off.
|
|
if code in known and not refresh:
|
|
skipped += 1
|
|
if progress:
|
|
progress(
|
|
SyncResult(discovered, downloaded, skipped, failed),
|
|
client.total_documents,
|
|
)
|
|
continue
|
|
try:
|
|
document = client.document(code)
|
|
checksum = archive_document(output, document)
|
|
fetched_at = datetime.now(timezone.utc).isoformat()
|
|
with connection:
|
|
connection.execute(
|
|
"""
|
|
INSERT INTO documents (code, source_sha256, fetched_at)
|
|
VALUES (?, ?, ?)
|
|
ON CONFLICT(code) DO UPDATE SET
|
|
source_sha256 = excluded.source_sha256,
|
|
fetched_at = excluded.fetched_at
|
|
""",
|
|
(int(document["Code"]), checksum, fetched_at),
|
|
)
|
|
connection.execute("DELETE FROM errors WHERE code = ?", (code,))
|
|
downloaded += 1
|
|
except Exception as error: # Continue the long-running archive after one bad record.
|
|
with connection:
|
|
connection.execute(
|
|
"""
|
|
INSERT INTO errors (code, message, failed_at) VALUES (?, ?, ?)
|
|
ON CONFLICT(code) DO UPDATE SET
|
|
message = excluded.message,
|
|
failed_at = excluded.failed_at
|
|
""",
|
|
(code, str(error), datetime.now(timezone.utc).isoformat()),
|
|
)
|
|
failed += 1
|
|
if progress:
|
|
progress(
|
|
SyncResult(discovered, downloaded, skipped, failed),
|
|
client.total_documents,
|
|
)
|
|
if not progress and discovered % 100 == 0:
|
|
LOGGER.info(
|
|
"discovered=%s downloaded=%s skipped=%s failed=%s",
|
|
discovered,
|
|
downloaded,
|
|
skipped,
|
|
failed,
|
|
)
|
|
connection.close()
|
|
return SyncResult(discovered, downloaded, skipped, failed)
|
|
|
|
|
|
def parse_args() -> argparse.Namespace:
|
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
parser.add_argument("--output", type=Path, default=Path("data/minjust-cbd"))
|
|
parser.add_argument("--limit", type=int, help="download only the first N documents")
|
|
parser.add_argument("--refresh", action="store_true", help="redownload known documents")
|
|
parser.add_argument("--requests-per-second", type=float, default=1)
|
|
parser.add_argument("--timeout", type=int, default=60)
|
|
parser.add_argument("--retries", type=int, default=5)
|
|
parser.add_argument("--version", action="version", version=APP_VERSION)
|
|
return parser.parse_args()
|
|
|
|
|
|
def main() -> int:
|
|
arguments = parse_args()
|
|
if arguments.limit is not None and arguments.limit <= 0:
|
|
raise SystemExit("--limit must be greater than zero")
|
|
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
|
|
result = sync_archive(
|
|
arguments.output,
|
|
CbdClient(
|
|
requests_per_second=arguments.requests_per_second,
|
|
timeout=arguments.timeout,
|
|
retries=arguments.retries,
|
|
),
|
|
arguments.limit,
|
|
arguments.refresh,
|
|
terminal_progress() if sys.stderr.isatty() else None,
|
|
)
|
|
print(
|
|
f"discovered={result.discovered} downloaded={result.downloaded} "
|
|
f"skipped={result.skipped} failed={result.failed}\n"
|
|
f"Akyldash Backend v{APP_VERSION} · Frontend — not created"
|
|
)
|
|
return int(result.failed > 0)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|