from __future__ import annotations import argparse import csv from decimal import Decimal, InvalidOperation import hashlib import io import json import os from dataclasses import dataclass from datetime import datetime, timedelta, timezone from pathlib import Path import posixpath import re import shutil import sys import time from typing import Protocol from urllib.error import HTTPError, URLError from urllib.parse import quote, urlencode from urllib.request import Request, urlopen import warnings import zipfile VENDOR_DIR = Path(__file__).resolve().parent / "vendor" sys.path.insert(0, str(VENDOR_DIR)) from openpyxl import load_workbook try: from runtime_config import METRIX_API_BASE_URL, METRIX_API_TOKEN, METRIX_DATABASE_CONNECTION_ID except ImportError: METRIX_API_BASE_URL = "http://188.5.127.115:18271" METRIX_API_TOKEN = "" METRIX_DATABASE_CONNECTION_ID = "" EXPECTED_TYPES = ( "5G下FDD干扰监控", "5G干扰监控", "700M下FDD干扰监控", "700M干扰监控", "SDR_FDD干扰监控", "SDR_TDD干扰监控", "反开RD干扰监控", ) DEFAULT_STORAGE_ID = "stg_4d9a910d72" DEFAULT_ROOT = "/网优日常优化数据文档/(勿删)干扰定时小时指标" CELL_DATA_DIRECTORIES = ( "/网优日常优化数据文档/日常性能报表/2026年/5G/5G小区信息表", "/网优日常优化数据文档/日常性能报表/2026年/700M/700M小区信息表", "/网优日常优化数据文档/日常性能报表/2026年/5G_反开/反开小区信息表", "/网优日常优化数据文档/日常性能报表/2026年/TDD_LTE/LTE基础信息数据/LTE_小区信息表", "/网优日常优化数据文档/日常性能报表/2026年/FDD_LTE/FDD基础信息数据/FDD小区信息表", ) FILE_RE = re.compile(r"^(?P.+)_LWP_每小时_过滤110_(?P\d{16})\.zip$", re.IGNORECASE) CELL_DATA_FILE_RE = re.compile(r"(?P20\d{6})\.xlsx$", re.IGNORECASE) LTE_PLMN = "460-00" DATABASE_NAME = "interference_etl" DATABASE_TABLE = "interference_hourly_summary" HEADER_FDD = ( "开始时间", "粒度", "子网ID", "子网名称", "网元ID", "管理网元", "eNodeB CUID", "eNodeB CU名称", "LTEID", "LTE名称", "E-UTRAN FDD小区ID", "E-UTRAN FDD小区名称", "cellId", "eNodeBId", "载波平均噪声干扰(dBm)", "集团-上下行总业务量(GB)", "RRC连接建立最大用户数", ) HEADER_NR = ( "开始时间", "粒度", "子网ID", "子网名称", "网元ID", "管理网元", "gNB CU-CP功能配置ID", "gNB CU-CP功能配置名称", "CU小区配置ID", "CU小区配置名称", "cellId", "duMeMoId", "gNBId", "gNBIdLength", "gNBplmn", "masterOperatorId", "nrCarrierGroupId", "nrPhysicalCellDUId", "小区上行平均干扰电平(dBm)", "5G上下行总流量(上行PDCP PDU数据量+下行PDCP成功发送数据量)(GB)", "RRC连接最大连接用户数", ) HEADER_SDR = ( "开始时间", "粒度", "子网ID", "子网名称", "网元ID", "管理网元", "eNodeBID", "eNodeB名称", "小区ID", "小区名称", "载波平均噪声干扰(dBm)", "集团-上下行总业务量(GB)", "RRC连接建立最大用户数", ) HEADER_RD = ( "开始时间", "粒度", "子网ID", "子网名称", "网元ID", "管理网元", "eNodeB CUID", "eNodeB CU名称", "LTEID", "LTE名称", "E-UTRAN TDD小区ID", "E-UTRAN TDD小区名称", "cellId", "eNodeBId", "载波平均噪声干扰(dBm)", "上下行总业务量(GB)", "RRC连接建立最大用户数", ) EXPECTED_HEADERS = { "5G下FDD干扰监控": HEADER_FDD, "5G干扰监控": HEADER_NR, "700M下FDD干扰监控": HEADER_FDD, "700M干扰监控": HEADER_NR, "SDR_FDD干扰监控": HEADER_SDR, "SDR_TDD干扰监控": HEADER_SDR, "反开RD干扰监控": HEADER_RD, } SUMMARY_HEADER = ( "hour_start", "hour_end", "cgi", "cell_name", "interference_dbm", "longitude", "latitude", ) class ProcessingError(RuntimeError): pass @dataclass(frozen=True) class Candidate: source_type: str window: str path: str size: int = 0 class Source(Protocol): def candidates(self, lookback_days: int) -> list[Candidate]: ... def download(self, path: str) -> bytes: ... def cell_data_workbooks(self) -> list[tuple[str, bytes]]: ... class SummaryStore(Protocol): def replace_latest(self, rows: list[dict[str, str]]) -> dict[str, object]: ... class MetrixApiClient: def __init__(self, base_url: str, token: str) -> None: if not token: raise ProcessingError("METRIX_API_TOKEN is required for Metrix API access") self.base_url = base_url.rstrip("/") self.token = token def get_bytes(self, endpoint: str, query: dict[str, object] | None = None, timeout: int = 30) -> bytes: return self._request("GET", endpoint, query=query, timeout=timeout) def get_json(self, endpoint: str, query: dict[str, object] | None = None, timeout: int = 30) -> object: return self._decode_json(self.get_bytes(endpoint, query, timeout)) def post_json(self, endpoint: str, payload: dict[str, object], timeout: int = 30) -> object: body = json.dumps(payload, ensure_ascii=False).encode("utf-8") return self._decode_json(self._request("POST", endpoint, body=body, timeout=timeout)) def _request( self, method: str, endpoint: str, query: dict[str, object] | None = None, body: bytes | None = None, timeout: int = 30, ) -> bytes: url = f"{self.base_url}{endpoint}" if query: url = f"{url}?{urlencode(query)}" headers = {"Authorization": f"Bearer {self.token}", "Accept": "application/json"} if body is not None: headers["Content-Type"] = "application/json" request = Request(url, data=body, headers=headers, method=method) last_error: Exception | None = None for attempt in range(3): try: with urlopen(request, timeout=timeout) as response: return response.read() except HTTPError as exc: detail = exc.read().decode("utf-8", "replace")[:1000] if exc.code < 500: raise ProcessingError(f"Metrix API {exc.code}: {detail}") from exc last_error = exc except (URLError, TimeoutError) as exc: last_error = exc if attempt < 2: time.sleep(2**attempt) raise ProcessingError(f"Metrix API request failed: {last_error}") @staticmethod def _decode_json(payload: bytes) -> object: try: return json.loads(payload.decode("utf-8")) except (UnicodeDecodeError, json.JSONDecodeError) as exc: raise ProcessingError("Metrix API returned invalid JSON") from exc class ApiSummaryStore: def __init__(self, client: MetrixApiClient, connection_id: str) -> None: if not connection_id: raise ProcessingError("METRIX_DATABASE_CONNECTION_ID is required for database output") self.client = client self.connection_id = connection_id self.table = DATABASE_TABLE def replace_latest(self, rows: list[dict[str, str]]) -> dict[str, object]: if not rows: raise ProcessingError("Refusing to replace database data with an empty batch") metric_times = {datetime.strptime(row["hour_start"], "%Y-%m-%d %H:%M:%S") for row in rows} if len(metric_times) != 1: raise ProcessingError("Database batch must contain exactly one metric hour") metric_time = next(iter(metric_times)) self._query( f"CREATE DATABASE IF NOT EXISTS `{DATABASE_NAME}` " "CHARACTER SET utf8mb4 COLLATE utf8mb4_unicode_ci" ) self._query( f""" CREATE TABLE IF NOT EXISTS `{self.table}` ( metric_time DATETIME NOT NULL COMMENT '指标开始时间', cgi VARCHAR(128) NOT NULL, cell_name VARCHAR(255) NOT NULL, interference_dbm DECIMAL(10,3) NOT NULL, longitude DECIMAL(10,6) NULL, latitude DECIMAL(10,6) NULL, PRIMARY KEY (metric_time, cgi) ) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4 """, database=DATABASE_NAME, ) columns = self.client.get_json( self._endpoint("columns"), {"database": DATABASE_NAME, "table": self.table}, ) if not isinstance(columns, list): raise ProcessingError("Metrix Database API returned invalid column metadata") existing_columns = {str(item.get("name")) for item in columns if isinstance(item, dict)} for column in ("longitude", "latitude"): if column not in existing_columns: self._query( f"ALTER TABLE `{self.table}` ADD COLUMN `{column}` DECIMAL(10,6) NULL", database=DATABASE_NAME, ) latest_payload = self._query( f"SELECT MAX(metric_time) AS latest_time FROM `{self.table}`", database=DATABASE_NAME, ) latest_rows = latest_payload.get("rows", []) latest_value = latest_rows[0].get("latest_time") if latest_rows else None latest_time = datetime.fromisoformat(str(latest_value)) if latest_value else None if latest_time is not None and latest_time > metric_time: raise ProcessingError( f"Database already contains newer metric time {latest_time:%Y-%m-%d %H:%M:%S}; " f"refusing to replace it with {metric_time:%Y-%m-%d %H:%M:%S}" ) metric_literal = f"'{metric_time:%Y-%m-%d %H:%M:%S}'" values = ",\n".join(self._row_values(metric_literal, row) for row in rows) script_payload = self.client.post_json( self._endpoint("run-script"), { "content": f""" START TRANSACTION; DELETE FROM `{self.table}` WHERE metric_time = {metric_literal}; INSERT INTO `{self.table}` (metric_time, cgi, cell_name, interference_dbm, longitude, latitude) VALUES {values}; DELETE FROM `{self.table}` WHERE metric_time <> {metric_literal}; COMMIT; """, "database": DATABASE_NAME, "stop_on_error": True, "single_session": True, }, timeout=120, ) if not isinstance(script_payload, dict) or not isinstance(script_payload.get("results"), list): raise ProcessingError("Metrix Database API returned an invalid script result") results = script_payload["results"] failed = next((item for item in results if not item.get("ok")), None) if failed is not None: raise ProcessingError(f"Database script failed at statement {failed.get('index')}: {failed.get('message', '')}") if len(results) != 5: raise ProcessingError(f"Database script returned {len(results)} results; expected 5") return { "enabled": True, "table": self.table, "metric_time": metric_time.strftime("%Y-%m-%d %H:%M:%S"), "inserted_rows": int(results[2].get("affected_rows") or 0), "refreshed_rows": int(results[1].get("affected_rows") or 0), "old_rows_deleted": int(results[3].get("affected_rows") or 0), } def _endpoint(self, action: str) -> str: return f"/api/databases/{quote(self.connection_id, safe='')}/{action}" def _query(self, sql: str, database: str = "") -> dict[str, object]: payload = self.client.post_json( self._endpoint("query"), {"sql": sql, "database": database, "page": 1, "page_size": 100}, timeout=120, ) if not isinstance(payload, dict): raise ProcessingError("Metrix Database API returned an invalid query result") return payload @staticmethod def _row_values(metric_literal: str, row: dict[str, str]) -> str: return "(" + ", ".join( ( metric_literal, sql_text_literal(row["cgi"]), sql_text_literal(row["cell_name"]), sql_decimal_literal(row["interference_dbm"], "interference_dbm"), sql_decimal_literal(row["longitude"], "longitude", nullable=True), sql_decimal_literal(row["latitude"], "latitude", nullable=True), ) ) + ")" class ApiSource: def __init__(self, client: MetrixApiClient, storage_id: str, root: str) -> None: self.client = client self.storage_id = storage_id self.root = root.rstrip("/") or "/" def candidates(self, lookback_days: int) -> list[Candidate]: root_entries = self._list_dir(self.root) date_dirs = sorted( (item for item in root_entries if item.get("is_dir") and re.fullmatch(r"\d{4}-\d{2}-\d{2}", item.get("name", ""))), key=lambda item: item["name"], ) selected_dirs = date_dirs[-max(1, lookback_days) :] result: list[Candidate] = [] for directory in selected_dirs: for item in self._list_dir(directory["path"]): if item.get("is_dir"): continue candidate = parse_candidate(item["path"], int(item.get("size") or 0)) if candidate is not None: result.append(candidate) return result def download(self, path: str) -> bytes: endpoint = f"/api/storages/{quote(self.storage_id, safe='')}/download" return self.client.get_bytes(endpoint, {"path": path}, timeout=120) def cell_data_workbooks(self) -> list[tuple[str, bytes]]: workbooks: list[tuple[str, bytes]] = [] for directory in CELL_DATA_DIRECTORIES: item = select_latest_cell_data_file(self._list_dir(directory), directory) path = str(item["path"]) workbooks.append((path, self.download(path))) return workbooks def _list_dir(self, path: str) -> list[dict[str, object]]: endpoint = f"/api/storages/{quote(self.storage_id, safe='')}/files" payload = self.client.get_json(endpoint, {"path": path, "recursive": "false"}) if not isinstance(payload, dict) or not isinstance(payload.get("entries"), list): raise ProcessingError("Metrix Storage API returned an invalid file list") return payload["entries"] class LocalSource: def __init__(self, root: Path) -> None: self.root = root.resolve() if not self.root.is_dir(): raise ProcessingError(f"Local source directory does not exist: {self.root}") def candidates(self, lookback_days: int) -> list[Candidate]: del lookback_days result: list[Candidate] = [] for path in self.root.rglob("*.zip"): candidate = parse_candidate(str(path.resolve()), path.stat().st_size) if candidate is not None: result.append(candidate) return result def download(self, path: str) -> bytes: return Path(path).read_bytes() def cell_data_workbooks(self) -> list[tuple[str, bytes]]: directories = [self.root / Path(directory.lstrip("/")) for directory in CELL_DATA_DIRECTORIES] existing = [directory for directory in directories if directory.is_dir()] if not existing: return [] if len(existing) != len(directories): raise ProcessingError("Local CellData source must contain all five configured directories") workbooks: list[tuple[str, bytes]] = [] for directory in directories: entries = [ {"name": path.name, "path": str(path.resolve()), "is_dir": path.is_dir()} for path in directory.iterdir() ] item = select_latest_cell_data_file(entries, str(directory)) path = str(item["path"]) workbooks.append((path, self.download(path))) return workbooks def parse_candidate(path: str, size: int = 0) -> Candidate | None: name = posixpath.basename(path.replace("\\", "/")) match = FILE_RE.fullmatch(name) if not match or match.group("source_type") not in EXPECTED_TYPES: return None return Candidate(match.group("source_type"), match.group("window"), path, size) def select_latest_cell_data_file(entries: list[dict[str, object]], directory: str) -> dict[str, object]: dated_files: list[tuple[datetime, str, dict[str, object]]] = [] for item in entries: if item.get("is_dir"): continue name = str(item.get("name") or "") match = CELL_DATA_FILE_RE.search(name) if not match: continue try: file_date = datetime.strptime(match.group("date"), "%Y%m%d") except ValueError: continue dated_files.append((file_date, name, item)) if not dated_files: raise ProcessingError(f"No dated CellData XLSX found in {directory}") return max(dated_files, key=lambda entry: (entry[0], entry[1]))[2] def parse_cell_data_workbook(raw_xlsx: bytes, path: str) -> dict[str, tuple[str, str]]: try: with warnings.catch_warnings(): warnings.filterwarnings("ignore", message="Workbook contains no default style") workbook = load_workbook(io.BytesIO(raw_xlsx), read_only=True, data_only=True) except Exception as exc: raise ProcessingError(f"Invalid CellData workbook {path}: {exc}") from exc try: if "小区信息表" not in workbook.sheetnames: raise ProcessingError(f"CellData workbook does not contain 小区信息表: {path}") rows = workbook["小区信息表"].iter_rows(values_only=True) try: header = tuple(normalize_cell(value) for value in next(rows)) except StopIteration as exc: raise ProcessingError(f"CellData workbook is empty: {path}") from exc required_columns = ("eNB/gNB", "CI", "经度", "纬度") missing = [column for column in required_columns if column not in header] if missing: raise ProcessingError(f"CellData workbook missing columns {missing}: {path}") indexes = {column: header.index(column) for column in required_columns} coordinates: dict[str, tuple[str, str]] = {} for row_number, row in enumerate(rows, start=2): node = normalize_cell(row[indexes["eNB/gNB"]]) cell = normalize_cell(row[indexes["CI"]]) longitude = normalize_cell(row[indexes["经度"]]) latitude = normalize_cell(row[indexes["纬度"]]) if not node or not cell or not longitude or not latitude: continue cgi = f"{LTE_PLMN}-{node}-{cell}" value = (longitude, latitude) existing = coordinates.get(cgi) if existing is not None and existing != value: raise ProcessingError(f"Conflicting CellData coordinates for {cgi} in {path}, row {row_number}") coordinates[cgi] = value return coordinates finally: workbook.close() def load_cell_coordinates(workbooks: list[tuple[str, bytes]]) -> dict[str, tuple[str, str]]: coordinates: dict[str, tuple[str, str]] = {} for path, raw_xlsx in workbooks: for cgi, value in parse_cell_data_workbook(raw_xlsx, path).items(): existing = coordinates.get(cgi) if existing is not None and existing != value: raise ProcessingError(f"Conflicting CellData coordinates for {cgi} across workbooks") coordinates[cgi] = value return coordinates def select_window(candidates: list[Candidate], requested: str = "") -> tuple[str, dict[str, Candidate], list[str]]: grouped: dict[str, dict[str, Candidate]] = {} duplicates: list[str] = [] for candidate in candidates: window_group = grouped.setdefault(candidate.window, {}) if candidate.source_type in window_group: duplicates.append(f"{candidate.window}/{candidate.source_type}") window_group[candidate.source_type] = candidate if duplicates: raise ProcessingError(f"Duplicate source files: {', '.join(sorted(duplicates))}") complete = sorted(window for window, items in grouped.items() if all(name in items for name in EXPECTED_TYPES)) if requested: if requested not in complete: present = sorted(grouped.get(requested, {})) missing = [name for name in EXPECTED_TYPES if name not in present] raise ProcessingError(f"Requested window is incomplete: {requested}; missing={missing}") selected = requested elif complete: selected = complete[-1] else: raise ProcessingError("No hour contains all seven interference source types") warnings_out: list[str] = [] latest_seen = max(grouped) if grouped else "" if latest_seen and latest_seen != selected: missing = [name for name in EXPECTED_TYPES if name not in grouped[latest_seen]] warnings_out.append(f"Latest observed window {latest_seen} is incomplete; using {selected}; missing={missing}") return selected, grouped[selected], warnings_out def parse_workbook(raw_zip: bytes, source_type: str) -> tuple[tuple[str, ...], list[tuple[object, ...]], str]: try: with zipfile.ZipFile(io.BytesIO(raw_zip)) as archive: bad_member = archive.testzip() if bad_member: raise ProcessingError(f"ZIP CRC check failed: {bad_member}") xlsx_members = [item for item in archive.infolist() if not item.is_dir() and item.filename.lower().endswith(".xlsx")] if len(xlsx_members) != 1: raise ProcessingError(f"Expected one XLSX member, found {len(xlsx_members)}") member = xlsx_members[0] workbook_bytes = archive.read(member) except zipfile.BadZipFile as exc: raise ProcessingError("Invalid ZIP archive") from exc with warnings.catch_warnings(): warnings.filterwarnings("ignore", message="Workbook contains no default style") workbook = load_workbook(io.BytesIO(workbook_bytes), read_only=True, data_only=True) try: if "Sheet0" not in workbook.sheetnames: raise ProcessingError("Workbook does not contain Sheet0") rows = workbook["Sheet0"].iter_rows(values_only=True) try: header = tuple(normalize_cell(value) for value in next(rows)) except StopIteration as exc: raise ProcessingError("Sheet0 is empty") from exc expected = EXPECTED_HEADERS[source_type] if header != expected: raise ProcessingError(f"Unexpected Sheet0 header for {source_type}: {header}") data = [tuple(row) for row in rows if any(value is not None and normalize_cell(value) != "" for value in row)] return header, data, member.filename finally: workbook.close() def process( source: Source, output_root: Path, lookback_days: int, requested_window: str = "", store: SummaryStore | None = None, ) -> Path: candidates = source.candidates(lookback_days) window, selected, warnings_out = select_window(candidates, requested_window) cell_data_workbooks = source.cell_data_workbooks() cell_coordinates = load_cell_coordinates(cell_data_workbooks) temp_dir = output_root.resolve() / f".{window}.tmp-{os.getpid()}" final_dir = output_root.resolve() / window ensure_scoped(output_root.resolve(), temp_dir) if temp_dir.exists(): shutil.rmtree(temp_dir) converted_dir = temp_dir / "converted" converted_dir.mkdir(parents=True) summary_rows: list[dict[str, str]] = [] manifest_files: list[dict[str, object]] = [] database_result: dict[str, object] = {"enabled": False} try: for source_type in EXPECTED_TYPES: candidate = selected[source_type] raw_zip = source.download(candidate.path) header, rows, member_name = parse_workbook(raw_zip, source_type) csv_name = f"{source_type}_{window}.csv" write_csv(converted_dir / csv_name, header, rows) for row_number, row in enumerate(rows, start=2): record = dict(zip(header, row, strict=True)) summary_rows.append( summary_record(record, source_type, candidate.path, window, row_number, cell_coordinates) ) manifest_files.append( { "source_size": candidate.size, "sha256": hashlib.sha256(raw_zip).hexdigest(), "xlsx_member": member_name, "rows": len(rows), "converted_csv": f"converted/{csv_name}", } ) summary_rows.sort(key=lambda item: (item["cgi"], item["cell_name"])) summary_name = f"interference_summary_{window}.csv" write_dict_csv(temp_dir / summary_name, SUMMARY_HEADER, summary_rows) if store is not None: database_result = store.replace_latest(summary_rows) matched_coordinates = sum(bool(row["longitude"] and row["latitude"]) for row in summary_rows) manifest = { "generated_at": datetime.now(timezone.utc).isoformat(), "window": window, "hour_start": window_bounds(window)[0], "hour_end": window_bounds(window)[1], "source_file_count": len(manifest_files), "cell_data_file_count": len(cell_data_workbooks), "summary_rows": len(summary_rows), "coordinate_matched_rows": matched_coordinates, "coordinate_unmatched_rows": len(summary_rows) - matched_coordinates, "warnings": warnings_out, "files": manifest_files, "summary_csv": summary_name, "database": database_result, } (temp_dir / "manifest.json").write_text(json.dumps(manifest, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") output_root.resolve().mkdir(parents=True, exist_ok=True) if final_dir.exists(): ensure_scoped(output_root.resolve(), final_dir) shutil.rmtree(final_dir) temp_dir.replace(final_dir) except Exception: shutil.rmtree(temp_dir, ignore_errors=True) raise print(f"selected_window={window}") print(f"source_files={len(manifest_files)}") print(f"cell_data_files={len(cell_data_workbooks)}") print(f"summary_rows={len(summary_rows)}") print(f"coordinate_matched_rows={matched_coordinates}") print(f"coordinate_unmatched_rows={len(summary_rows) - matched_coordinates}") if database_result["enabled"]: print(f"database_metric_time={database_result['metric_time']}") print(f"database_inserted_rows={database_result['inserted_rows']}") print(f"database_old_rows_deleted={database_result['old_rows_deleted']}") for message in warnings_out: print(f"warning={message}") print(f"output={final_dir}") return final_dir def summary_record( record: dict[str, object], source_type: str, source_path: str, window: str, row_number: int, cell_coordinates: dict[str, tuple[str, str]] | None = None, ) -> dict[str, str]: hour_start = normalize_cell(record["开始时间"]) expected_start, hour_end = window_bounds(window) if hour_start != expected_start: raise ProcessingError(f"{source_path}: row {row_number} time {hour_start!r} does not match {expected_start!r}") if source_type in ("5G干扰监控", "700M干扰监控"): cgi = build_cgi("", record, ("gNBplmn", "gNBId", "cellId"), source_path, row_number) cell_name = required(record, "CU小区配置名称", source_path, row_number) interference = required(record, "小区上行平均干扰电平(dBm)", source_path, row_number) elif source_type in ("SDR_FDD干扰监控", "SDR_TDD干扰监控"): cgi = build_cgi(LTE_PLMN, record, ("eNodeBID", "小区ID"), source_path, row_number) cell_name = required(record, "小区名称", source_path, row_number) interference = required(record, "载波平均噪声干扰(dBm)", source_path, row_number) elif source_type == "反开RD干扰监控": cgi = build_cgi(LTE_PLMN, record, ("eNodeBId", "cellId"), source_path, row_number) cell_name = required(record, "E-UTRAN TDD小区名称", source_path, row_number) interference = required(record, "载波平均噪声干扰(dBm)", source_path, row_number) else: cgi = build_cgi(LTE_PLMN, record, ("eNodeBId", "cellId"), source_path, row_number) cell_name = required(record, "E-UTRAN FDD小区名称", source_path, row_number) interference = required(record, "载波平均噪声干扰(dBm)", source_path, row_number) longitude, latitude = (cell_coordinates or {}).get(cgi, ("", "")) return { "hour_start": hour_start, "hour_end": hour_end, "cgi": cgi, "cell_name": cell_name, "interference_dbm": interference, "longitude": longitude, "latitude": latitude, } def build_cgi(prefix: str, record: dict[str, object], columns: tuple[str, ...], path: str, row: int) -> str: parts = [required(record, column, path, row) for column in columns] return "-".join(([prefix] if prefix else []) + parts) def required(record: dict[str, object], column: str, path: str, row: int) -> str: value = normalize_cell(record[column]) if not value: raise ProcessingError(f"{path}: row {row} has empty {column}") return value def window_bounds(window: str) -> tuple[str, str]: if not re.fullmatch(r"\d{16}", window): raise ProcessingError(f"Invalid window: {window}") start = datetime.strptime(window[:12], "%Y%m%d%H%M") end_hour = int(window[12:14]) end_minute = int(window[14:16]) end = start.replace(hour=end_hour, minute=end_minute) if end <= start: end += timedelta(days=1) return start.strftime("%Y-%m-%d %H:%M:%S"), end.strftime("%Y-%m-%d %H:%M:%S") def normalize_cell(value: object) -> str: if value is None: return "" if isinstance(value, datetime): return value.strftime("%Y-%m-%d %H:%M:%S") if isinstance(value, float) and value.is_integer(): return str(int(value)) return str(value).strip() def sql_text_literal(value: str) -> str: encoded = value.encode("utf-8").hex() return "''" if not encoded else f"CONVERT(0x{encoded} USING utf8mb4)" def sql_decimal_literal(value: str, field: str, nullable: bool = False) -> str: normalized = value.strip() if not normalized and nullable: return "NULL" try: number = Decimal(normalized) except InvalidOperation as exc: raise ProcessingError(f"Invalid decimal value for {field}: {value}") from exc if not number.is_finite(): raise ProcessingError(f"Invalid decimal value for {field}: {value}") return format(number, "f") def write_csv(path: Path, header: tuple[str, ...], rows: list[tuple[object, ...]]) -> None: with path.open("w", encoding="utf-8-sig", newline="") as file: writer = csv.writer(file) writer.writerow(header) for row in rows: writer.writerow(normalize_cell(value) for value in row) def write_dict_csv(path: Path, header: tuple[str, ...], rows: list[dict[str, str]]) -> None: with path.open("w", encoding="utf-8-sig", newline="") as file: writer = csv.DictWriter(file, fieldnames=header, extrasaction="raise") writer.writeheader() writer.writerows(rows) def ensure_scoped(root: Path, target: Path) -> None: if target == root or root not in target.parents: raise ProcessingError(f"Refusing to modify path outside output root: {target}") def build_parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser(description="Process the latest complete hour of interference KPI files") parser.add_argument("--source-dir", type=Path, help="Use a local source tree instead of the Metrix storage API") parser.add_argument("--output-dir", type=Path, default=Path(os.getenv("INTERFERENCE_OUTPUT_DIR", "output"))) parser.add_argument("--window", default=os.getenv("INTERFERENCE_WINDOW", ""), help="Optional exact 16-digit source window") parser.add_argument("--lookback-days", type=int, default=int(os.getenv("INTERFERENCE_LOOKBACK_DAYS", "3"))) parser.add_argument("--storage-id", default=os.getenv("METRIX_STORAGE_ID", DEFAULT_STORAGE_ID)) parser.add_argument("--root", default=os.getenv("INTERFERENCE_SOURCE_ROOT", DEFAULT_ROOT)) parser.add_argument("--no-database", action="store_true", help="Generate files without writing the database") return parser def main(argv: list[str] | None = None) -> int: args = build_parser().parse_args(argv) if args.lookback_days < 1: raise ProcessingError("lookback-days must be at least 1") client: MetrixApiClient | None = None if args.source_dir: source: Source = LocalSource(args.source_dir) else: client = MetrixApiClient(METRIX_API_BASE_URL, METRIX_API_TOKEN) source = ApiSource(client, args.storage_id, args.root) if args.no_database: store = None else: client = client or MetrixApiClient(METRIX_API_BASE_URL, METRIX_API_TOKEN) store = ApiSummaryStore(client, METRIX_DATABASE_CONNECTION_ID) process(source, args.output_dir, args.lookback_days, args.window, store=store) return 0 if __name__ == "__main__": try: raise SystemExit(main()) except ProcessingError as exc: print(f"error={exc}", file=sys.stderr) raise SystemExit(1) from exc