198 lines
14 KiB
Python
198 lines
14 KiB
Python
#!/usr/bin/env python3
|
||
"""Build a draft grouped view of the source GeneralClassifiers hierarchy."""
|
||
|
||
import argparse
|
||
import hashlib
|
||
import json
|
||
import re
|
||
import sqlite3
|
||
import unicodedata
|
||
from concurrent.futures import ThreadPoolExecutor
|
||
from contextlib import closing
|
||
from pathlib import Path
|
||
|
||
|
||
GROUPS = {
|
||
"constitutional_order": ("Конституционный строй и права", "Конституциялык түзүлүш жана укуктар"),
|
||
"government": ("Государство и публичное управление", "Мамлекет жана мамлекеттик башкаруу"),
|
||
"justice": ("Право, суд и правопорядок", "Укук, сот жана укук тартиби"),
|
||
"international": ("Международные отношения", "Эл аралык мамилелер"),
|
||
"security": ("Оборона и безопасность", "Коргонуу жана коопсуздук"),
|
||
"economy": ("Экономика, финансы и предпринимательство", "Экономика, финансы жана ишкердик"),
|
||
"labor_social": ("Труд и социальная защита", "Эмгек жана социалдык коргоо"),
|
||
"health": ("Здравоохранение", "Саламаттыкты сактоо"),
|
||
"education_culture": ("Образование, наука и культура", "Билим берүү, илим жана маданият"),
|
||
"land_environment": ("Земля, природные ресурсы и окружающая среда", "Жер, жаратылыш ресурстары жана айлана-чөйрө"),
|
||
"agriculture": ("Сельское хозяйство", "Айыл чарба"),
|
||
"infrastructure": ("Транспорт, связь, строительство и жильё", "Транспорт, байланыш, курулуш жана турак жай"),
|
||
"information": ("Информация и средства массовой информации", "Маалымат жана жалпыга маалымдоо каражаттары"),
|
||
"civil_family": ("Гражданские, семейные и жилищные отношения", "Жарандык, үй-бүлөлүк жана турак жай мамилелери"),
|
||
"cross_cutting": ("Общие и межотраслевые вопросы", "Жалпы жана тармактар аралык маселелер"),
|
||
"review_required": ("Требуется смысловая проверка", "Маанисин текшерүү талап кылынат"),
|
||
}
|
||
|
||
RULES = [
|
||
("international", r"международ|внешн(яя|ей) полит|дипломат|консул|иностранн|снг|содружеств|эл аралык"),
|
||
("security", r"оборон|военн|арм|мобилизац|государственн(ая|ой) безопасност|чрезвыча|гражданск(ая|ой) оборон|погранич"),
|
||
("health", r"здравоохран|медицин|лекарств|санитар|эпидеми|трансплантац|донорств|охрана здоровья"),
|
||
("education_culture", r"образован|просвещен|наук|культур|искусств|(?<![а-я])спорт|туризм|библиотек|музе|архив|издательств"),
|
||
("labor_social", r"труд|занятост|пенси|социальн|соцстрах|безработ|инвалид|пособи|охрана труда|семейн(ая|ые) поддержк"),
|
||
("land_environment", r"земл|природн(ые|ых) ресурс|окружающ|охрана природы|недр|водн(ые|ых) ресурс|атмосферн|животн(ый|ого) мир|лесн(ой|ое) хозяйств|эколог"),
|
||
("agriculture", r"сельск|аграр|растениевод|животновод|земледел|рыболов|рыбовод|зерновод|семеновод|ветеринар"),
|
||
("infrastructure", r"транспорт|связи|дорог|железнодорож|авиац|строительств|градостро|жилищ|коммунальн|энергетик|почтов|телекоммуникац"),
|
||
("information", r"информац|средств массовой информац|радио|телевиден|персональн(ые|ых) данные|средства массовой информации"),
|
||
("economy", r"финанс|бюджет|налог|кредит|эконом|предприним|хозяйственн|промышлен|торгов|тамож|внешнеэконом|банк|страхован|инвестиц|ценн(ые|ых) бумаг|лицензировани|валют|конкуренц|монопол|государственн(ая|ой) собственност"),
|
||
("justice", r"(?<![а-я])суд(?![а-я])|юстици|прокуратур|адвокат|нотариат|уголов|правонаруш|правопоряд|общественн(ого|ый) поряд|административн(ой|ая) ответственност|исполнени(е|я) наказан|следств|розыск|правоохран|систематизац.*норматив|законодательств о нормативн|арбитражн(ый|ого) процесс"),
|
||
("constitutional_order", r"конституц|государственн(ого|ый) стро|права и свобод|гражданств|выбор|референдум|парламент|жогорку кенеш|избирательн|символ|государственн(ый|ого) язык|законодательная деятельность"),
|
||
("government", r"государственн(ая|ой) служ|государственн(ое|ого) управлен|местн(ое|ого) самоуправлен|местная государственная администрация|орган(ы|ов) государственн|административн(ое|ого) управлен|государственн(ая|ой) власть|муниципальн|территориальн|государственн(ая|ой) наград|государственн(ый|ого) архив|министерств|кабинет министров|правительство|государственн(ого|ая) аппарата|нормотворческая деятельность|решения по кадровым|назначение на должность|освобождение от должности"),
|
||
("civil_family", r"гражданск|семейн|брачн|жилищ|собственност|наследован|договор|обязательств|авторск|интеллектуальн|потребител|личн(ые|ых) неимущественн|опек|попечительств"),
|
||
("cross_cutting", r"общ(ие|им) вопрос|межотрасл|общие положения|классификатор|учёт норматив|учет норматив|праздник|увековеч|по другим вопросам|^законодательство$"),
|
||
]
|
||
|
||
|
||
def clean(value):
|
||
return unicodedata.normalize("NFC", value.strip()) if isinstance(value, str) and value.strip() else None
|
||
|
||
|
||
def group_for(label):
|
||
text = (label or "").casefold()
|
||
matches = [code for code, pattern in RULES if re.search(pattern, text)]
|
||
return matches[0] if len(matches) == 1 else "review_required"
|
||
|
||
|
||
def node_id(path):
|
||
raw = json.dumps(path, ensure_ascii=False, separators=(",", ":"))
|
||
return "source-" + hashlib.sha256(raw.encode("utf-8")).hexdigest()[:16]
|
||
|
||
|
||
def add_nodes(nodes, classifiers, parent_path, seen, document_code):
|
||
if classifiers is None:
|
||
return
|
||
if not isinstance(classifiers, list):
|
||
raise RuntimeError(f"normalized document {document_code} contains a non-list GeneralClassifiers branch")
|
||
for source in classifiers:
|
||
if not isinstance(source, dict):
|
||
raise RuntimeError(f"normalized document {document_code} contains a non-object GeneralClassifiers node")
|
||
name = source.get("Name") if isinstance(source.get("Name"), dict) else {}
|
||
labels = {"ru": clean(name.get("Rus")), "ky": clean(name.get("Kyr"))}
|
||
code = source.get("Code")
|
||
identity = (str(code) if code is not None else None, labels["ru"], labels["ky"])
|
||
path = parent_path + [identity]
|
||
identifier = node_id(path)
|
||
if identifier not in nodes:
|
||
nodes[identifier] = {
|
||
"id": identifier,
|
||
"source_codes": set(),
|
||
"labels": labels,
|
||
"document_count": 0,
|
||
"children": {},
|
||
"_path": path,
|
||
}
|
||
node = nodes[identifier]
|
||
if code is not None:
|
||
node["source_codes"].add(str(code))
|
||
if identifier not in seen:
|
||
node["document_count"] += 1
|
||
seen.add(identifier)
|
||
add_nodes(nodes, source.get("GeneralClassifiers"), path, seen, document_code)
|
||
|
||
|
||
def emit_node(node, children):
|
||
result = {
|
||
"id": node["id"],
|
||
"source_codes": sorted(node["source_codes"]),
|
||
"labels": node["labels"],
|
||
"document_count": node["document_count"],
|
||
"children": children,
|
||
}
|
||
if "grouping_status" in node:
|
||
result["grouping_status"] = node["grouping_status"]
|
||
return result
|
||
|
||
|
||
def build(normalized_root):
|
||
db_path = normalized_root / "manifest.sqlite3"
|
||
with closing(sqlite3.connect(db_path)) as connection:
|
||
codes = [row[0] for row in connection.execute("SELECT code FROM documents WHERE state='success' ORDER BY code")]
|
||
nodes = {}
|
||
roots = {}
|
||
source_root = normalized_root / "documents"
|
||
processed = 0
|
||
|
||
def read(code):
|
||
path = source_root / code / "document.json"
|
||
try:
|
||
data = json.loads(path.read_text(encoding="utf-8"))
|
||
except (OSError, json.JSONDecodeError) as error:
|
||
raise RuntimeError(f"cannot read normalized document {code}: {error}") from error
|
||
if not isinstance(data, dict):
|
||
raise RuntimeError(f"normalized document {code} must contain a JSON object")
|
||
classifiers = data.get("general_classifiers")
|
||
if not isinstance(classifiers, list):
|
||
raise RuntimeError(f"normalized document {code} must contain a general_classifiers list")
|
||
return code, classifiers
|
||
|
||
with ThreadPoolExecutor(max_workers=24) as pool:
|
||
for offset in range(0, len(codes), 512):
|
||
for code, classifiers in pool.map(read, codes[offset:offset + 512]):
|
||
seen = set()
|
||
add_nodes(nodes, classifiers, [], seen, code)
|
||
for source in classifiers if isinstance(classifiers, list) else []:
|
||
if not isinstance(source, dict):
|
||
continue
|
||
name = source.get("Name") if isinstance(source.get("Name"), dict) else {}
|
||
labels = {"ru": clean(name.get("Rus")), "ky": clean(name.get("Kyr"))}
|
||
identity = (str(source.get("Code")) if source.get("Code") is not None else None, labels["ru"], labels["ky"])
|
||
key = node_id([identity])
|
||
roots.setdefault(key, nodes[key])
|
||
processed += 1
|
||
|
||
by_parent = {}
|
||
for node in nodes.values():
|
||
parent = node["_path"][:-1]
|
||
parent_id = node_id(parent) if parent else None
|
||
by_parent.setdefault(parent_id, []).append(node)
|
||
|
||
groups = {code: {"code": code, "labels": {"ru": names[0], "ky": names[1]}, "source_roots": []} for code, names in GROUPS.items()}
|
||
for root in roots.values():
|
||
code = group_for(root["labels"]["ru"])
|
||
root["grouping_status"] = "manual_review" if code == "review_required" else "provisional_lexical"
|
||
groups[code]["source_roots"].append(root)
|
||
|
||
def render(node):
|
||
children = sorted(by_parent.get(node["id"], []), key=lambda item: (item["labels"]["ru"] or "", item["labels"]["ky"] or "", item["id"]))
|
||
return emit_node(node, [render(child) for child in children])
|
||
|
||
missing_ky_count = sum(not node["labels"]["ky"] for node in nodes.values())
|
||
assigned_roots = [root["id"] for group in groups.values() for root in group["source_roots"]]
|
||
assert len(assigned_roots) == len(roots) == len(set(assigned_roots))
|
||
catalog = {
|
||
"status": "draft; semantic group assignment requires review",
|
||
"source": "Ministry of Justice GeneralClassifiers from normalized documents",
|
||
"source_document_count": processed,
|
||
"source_node_count": len(nodes),
|
||
"source_root_count": len(roots),
|
||
"nodes_missing_kyrgyz_label": missing_ky_count,
|
||
"id_policy": "provisional SHA-256 of source code and bilingual ancestor labels; null source codes are common",
|
||
"groups": [
|
||
{"code": code, "labels": group["labels"], "source_root_count": len(group["source_roots"]), "children": [render(root) for root in sorted(group["source_roots"], key=lambda item: (item["labels"]["ru"] or "", item["labels"]["ky"] or "", item["id"]))]}
|
||
for code, group in groups.items()
|
||
],
|
||
}
|
||
return catalog
|
||
|
||
|
||
def main():
|
||
parser = argparse.ArgumentParser()
|
||
parser.add_argument("normalized_root", type=Path)
|
||
parser.add_argument("output", type=Path)
|
||
args = parser.parse_args()
|
||
catalog = build(args.normalized_root)
|
||
args.output.parent.mkdir(parents=True, exist_ok=True)
|
||
args.output.write_text(json.dumps(catalog, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
||
print(f"documents={catalog['source_document_count']} nodes={catalog['source_node_count']} roots={catalog['source_root_count']} groups={len(catalog['groups'])} missing_ky={catalog['nodes_missing_kyrgyz_label']} review_roots={next(group['source_root_count'] for group in catalog['groups'] if group['code'] == 'review_required')}")
|
||
|
||
|
||
if __name__ == "__main__":
|
||
main()
|