Files
AWatch-rus/clickhouse-1c/etl/build_business_event_exports.py
T

349 lines
15 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
from __future__ import annotations
import argparse
import hashlib
import json
import os
import re
from dataclasses import dataclass
from datetime import UTC, datetime
from pathlib import Path
from typing import Any
from load_1c_exports import Config, iter_rows, load_config, normalize_ts
def parse_args() -> argparse.Namespace:
p = argparse.ArgumentParser(description="Build canonical business-event exports from read-only 1C file exports")
p.add_argument("--config", required=True, help="Path to YAML config")
return p.parse_args()
@dataclass(frozen=True)
class DocumentMeta:
infobase: str
company_entity_key: str
organization: str
department: str
document_id: str
document_number: str
document_type: str
registrar: str
user: str
counterparty: str
operation_type: str
amount: float
def collapse_ws(value: str) -> str:
return re.sub(r"\s+", " ", value).strip()
def normalize_base_path(value: str) -> str:
text = value.upper().replace("\\", "/")
text = re.sub(r"/+", "/", text)
text = re.sub(r"[^0-9A-ZА-ЯЁ:/._ -]+", " ", text)
return collapse_ws(text)
def normalize_infobase_key(value: str) -> str:
text = value.upper()
text = re.sub(r"(^|\s)20[0-9]{2}($|\s)", " ", text)
text = re.sub(r"[^0-9A-ZА-ЯЁ]+", " ", text)
return collapse_ws(text)
def canonical_company_entity_key(base_id: str = "", base_path: str = "", infobase: str = "") -> str:
if base_id:
return f"baseid:{base_id.strip()}"
if base_path:
normalized_path = normalize_base_path(base_path)
if normalized_path:
return f"basepath:{normalized_path}"
normalized_infobase = normalize_infobase_key(infobase)
return f"infobase:{normalized_infobase}" if normalized_infobase else ""
def source_format(conf: Config, dataset: str) -> str:
return conf.formats.get(dataset, conf.formats.get("default", "jsonl"))
def landing_root(conf: Config, dataset: str) -> Path:
return Path(conf.landing[dataset])
def file_ready(path: Path, conf: Config) -> bool:
age_seconds = max(0, int((datetime.now(UTC) - datetime.fromtimestamp(path.stat().st_mtime, UTC)).total_seconds()))
return age_seconds >= conf.min_file_age_seconds
def iter_dataset_files(conf: Config, dataset: str) -> list[Path]:
root = landing_root(conf, dataset)
if not root.exists():
return []
return [p for p in sorted(root.iterdir()) if p.is_file() and file_ready(p, conf)]
def stable_id(prefix: str, *parts: Any) -> str:
payload = "|".join(str(part or "") for part in parts)
digest = hashlib.sha1(payload.encode("utf-8"), usedforsecurity=False).hexdigest()[:16]
return f"{prefix}:{digest}"
def write_jsonl(path: Path, rows: list[dict[str, Any]], *, min_age_seconds: int) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
if not rows:
path.unlink(missing_ok=True)
return
tmp = path.with_suffix(path.suffix + ".tmp")
payload = "\n".join(json.dumps(row, ensure_ascii=False) for row in rows) + "\n"
tmp.write_text(payload, encoding="utf-8")
tmp.replace(path)
aged_mtime = max(0, int(datetime.now(UTC).timestamp()) - max(min_age_seconds + 1, 5))
os.utime(path, (aged_mtime, aged_mtime))
def load_company_index(conf: Config) -> dict[str, str]:
latest: dict[str, tuple[datetime, str]] = {}
for path in iter_dataset_files(conf, "companies"):
for row in iter_rows(path, source_format(conf, "companies")):
infobase = str(row.get("infobase", "")).strip()
if not infobase:
continue
entity_key = canonical_company_entity_key(
base_id=str(row.get("base_id", "")).strip(),
base_path=str(row.get("base_path", "")).strip(),
infobase=infobase,
)
ts = normalize_ts(row.get("ts"))
current = latest.get(infobase)
if current is None or ts >= current[0]:
latest[infobase] = (ts, entity_key)
return {infobase: entity_key for infobase, (_ts, entity_key) in latest.items()}
def derive_company_entity_key(row: dict[str, Any], company_index: dict[str, str]) -> str:
explicit = str(row.get("company_entity_key", "")).strip()
if explicit:
return explicit
infobase = str(row.get("infobase", "")).strip()
return company_index.get(infobase, "") or canonical_company_entity_key(
base_id=str(row.get("base_id", "")).strip(),
base_path=str(row.get("base_path", "")).strip(),
infobase=infobase,
)
def build_document_index(conf: Config, company_index: dict[str, str]) -> tuple[dict[tuple[str, str], DocumentMeta], dict[tuple[str, str], DocumentMeta]]:
rows: list[dict[str, Any]] = []
for path in iter_dataset_files(conf, "documents"):
rows.extend(iter_rows(path, source_format(conf, "documents")))
return build_document_index_from_rows(rows, company_index)
def build_document_index_from_rows(rows: list[dict[str, Any]], company_index: dict[str, str]) -> tuple[dict[tuple[str, str], DocumentMeta], dict[tuple[str, str], DocumentMeta]]:
by_id: dict[tuple[str, str], DocumentMeta] = {}
by_number: dict[tuple[str, str], DocumentMeta] = {}
for row in rows:
infobase = str(row.get("infobase", "")).strip()
document_id = str(row.get("doc_id", row.get("document_id", ""))).strip()
document_number = str(row.get("doc_number", row.get("document_number", ""))).strip()
if not infobase:
continue
meta = DocumentMeta(
infobase=infobase,
company_entity_key=derive_company_entity_key(row, company_index),
organization=str(row.get("organization", "")).strip(),
department=str(row.get("department", "")).strip(),
document_id=document_id,
document_number=document_number,
document_type=str(row.get("doc_type", row.get("document_type", ""))).strip(),
registrar=document_id or document_number,
user=str(row.get("author", row.get("user", ""))).strip(),
counterparty=str(row.get("counterparty", "")).strip(),
operation_type=str(row.get("operation_type", "")).strip(),
amount=float(row.get("amount", 0) or 0),
)
if document_id:
by_id[(infobase, document_id)] = meta
if document_number:
by_number[(infobase, document_number)] = meta
return by_id, by_number
def lookup_document_meta(
row: dict[str, Any],
by_id: dict[tuple[str, str], DocumentMeta],
by_number: dict[tuple[str, str], DocumentMeta],
) -> DocumentMeta | None:
infobase = str(row.get("infobase", "")).strip()
registrar = str(row.get("registrar", row.get("document_id", row.get("doc_id", "")))).strip()
document_number = str(row.get("document_number", row.get("doc_number", ""))).strip()
if infobase and registrar and (infobase, registrar) in by_id:
return by_id[(infobase, registrar)]
if infobase and document_number and (infobase, document_number) in by_number:
return by_number[(infobase, document_number)]
return None
def build_document_events(rows: list[dict[str, Any]], source_file: str, company_index: dict[str, str]) -> list[dict[str, Any]]:
events: list[dict[str, Any]] = []
for row in rows:
infobase = str(row.get("infobase", "")).strip()
document_id = str(row.get("doc_id", row.get("document_id", ""))).strip()
document_number = str(row.get("doc_number", row.get("document_number", ""))).strip()
document_type = str(row.get("doc_type", row.get("document_type", ""))).strip()
ts = normalize_ts(row.get("ts") or row.get("posted_at") or row.get("created_at"))
company_entity_key = derive_company_entity_key(row, company_index)
events.append(
{
"ts": ts.isoformat(),
"event_id": stable_id("document_snapshot", source_file, infobase, document_id, document_number, ts.isoformat()),
"infobase": infobase,
"company_entity_key": company_entity_key,
"organization": str(row.get("organization", "")).strip(),
"department": str(row.get("department", "")).strip(),
"document_id": document_id,
"document_number": document_number,
"document_type": document_type,
"registrar": document_id or document_number,
"operation_type": str(row.get("operation_type", "")).strip(),
"event_kind": "document_snapshot",
"user": str(row.get("author", row.get("user", ""))).strip(),
"counterparty": str(row.get("counterparty", "")).strip(),
"counterparty_inn": str(row.get("counterparty_inn", "")).strip(),
"debit_account": "",
"credit_account": "",
"amount": float(row.get("amount", 0) or 0),
"currency": str(row.get("currency", "RUB")).strip() or "RUB",
"line_no": 0,
"evidence_ref": f"document:{document_id or document_number}",
}
)
return events
def build_posting_events(
rows: list[dict[str, Any]],
source_file: str,
company_index: dict[str, str],
by_id: dict[tuple[str, str], DocumentMeta],
by_number: dict[tuple[str, str], DocumentMeta],
) -> list[dict[str, Any]]:
events: list[dict[str, Any]] = []
for idx, row in enumerate(rows, start=1):
meta = lookup_document_meta(row, by_id, by_number)
infobase = str(row.get("infobase", "")).strip()
registrar = str(row.get("registrar", "")).strip()
ts = normalize_ts(row.get("ts"))
company_entity_key = meta.company_entity_key if meta else derive_company_entity_key(row, company_index)
line_no = int(row.get("line_no", idx) or idx)
events.append(
{
"ts": ts.isoformat(),
"event_id": stable_id("posting", source_file, infobase, registrar, line_no, ts.isoformat()),
"infobase": infobase,
"company_entity_key": company_entity_key,
"organization": meta.organization if meta else str(row.get("organization", "")).strip(),
"department": meta.department if meta else str(row.get("department", "")).strip(),
"document_id": meta.document_id if meta else registrar,
"document_number": meta.document_number if meta else str(row.get("document_number", "")).strip(),
"document_type": meta.document_type if meta else str(row.get("document_type", "")).strip(),
"registrar": registrar,
"operation_type": str(row.get("operation_type", meta.operation_type if meta else "")).strip(),
"event_kind": "posting",
"user": meta.user if meta else str(row.get("user", row.get("author", ""))).strip(),
"counterparty": meta.counterparty if meta else str(row.get("counterparty", "")).strip(),
"counterparty_inn": str(row.get("counterparty_inn", "")).strip(),
"debit_account": str(row.get("account_dt", row.get("debit_account", ""))).strip(),
"credit_account": str(row.get("account_ct", row.get("credit_account", ""))).strip(),
"amount": float(row.get("amount", 0) or 0),
"currency": str(row.get("currency", "RUB")).strip() or "RUB",
"line_no": line_no,
"evidence_ref": f"posting:{registrar}:{line_no}",
}
)
return events
def build_document_changes(
rows: list[dict[str, Any]],
source_file: str,
company_index: dict[str, str],
by_id: dict[tuple[str, str], DocumentMeta],
by_number: dict[tuple[str, str], DocumentMeta],
) -> list[dict[str, Any]]:
changes: list[dict[str, Any]] = []
for row in rows:
infobase = str(row.get("infobase", "")).strip()
object_type = str(row.get("object_type", "")).strip()
object_id = str(row.get("object_id", "")).strip()
ts = normalize_ts(row.get("ts"))
meta = lookup_document_meta(
{
"infobase": infobase,
"registrar": row.get("document_id") or (object_id if object_type == "document" else ""),
"document_number": row.get("document_number", ""),
},
by_id,
by_number,
)
company_entity_key = meta.company_entity_key if meta else derive_company_entity_key(row, company_index)
document_id = meta.document_id if meta else (object_id if object_type == "document" else str(row.get("document_id", "")).strip())
changes.append(
{
"ts": ts.isoformat(),
"change_id": stable_id("change", source_file, infobase, object_type, object_id, row.get("action", ""), ts.isoformat()),
"infobase": infobase,
"company_entity_key": company_entity_key,
"organization": meta.organization if meta else str(row.get("organization", "")).strip(),
"document_id": document_id,
"document_number": meta.document_number if meta else str(row.get("document_number", "")).strip(),
"document_type": meta.document_type if meta else str(row.get("document_type", "")).strip(),
"change_kind": str(row.get("change_kind", row.get("action", object_type))).strip(),
"field_name": str(row.get("field_name", object_type)).strip(),
"user": str(row.get("user", row.get("author", ""))).strip(),
"before_value": str(row.get("before_value", row.get("before_hash", ""))).strip(),
"after_value": str(row.get("after_value", row.get("after_hash", ""))).strip(),
"risk_tag": str(row.get("risk_tag", "")).strip(),
"evidence_ref": f"audit:{object_type}:{object_id}",
}
)
return changes
def output_path(conf: Config, dataset: str, source_path: Path, prefix: str) -> Path:
return landing_root(conf, dataset) / f"{prefix}-{source_path.stem}.jsonl"
def main() -> int:
args = parse_args()
conf = load_config(args.config)
company_index = load_company_index(conf)
by_id, by_number = build_document_index(conf, company_index)
for path in iter_dataset_files(conf, "documents"):
rows = iter_rows(path, source_format(conf, "documents"))
out = output_path(conf, "business_events", path, "business-events-documents")
write_jsonl(out, build_document_events(rows, path.name, company_index), min_age_seconds=conf.min_file_age_seconds)
print(f"built business_events from documents: {path.name} rows={len(rows)}")
for path in iter_dataset_files(conf, "postings"):
rows = iter_rows(path, source_format(conf, "postings"))
out = output_path(conf, "business_events", path, "business-events-postings")
write_jsonl(out, build_posting_events(rows, path.name, company_index, by_id, by_number), min_age_seconds=conf.min_file_age_seconds)
print(f"built business_events from postings: {path.name} rows={len(rows)}")
for path in iter_dataset_files(conf, "audit"):
rows = iter_rows(path, source_format(conf, "audit"))
out = output_path(conf, "document_changes", path, "document-changes-audit")
write_jsonl(out, build_document_changes(rows, path.name, company_index, by_id, by_number), min_age_seconds=conf.min_file_age_seconds)
print(f"built document_changes from audit: {path.name} rows={len(rows)}")
return 0
if __name__ == "__main__":
raise SystemExit(main())