833 lines
34 KiB
Python
833 lines
34 KiB
Python
from __future__ import annotations
|
||
|
||
import argparse
|
||
import csv
|
||
from decimal import Decimal, InvalidOperation
|
||
import hashlib
|
||
import io
|
||
import json
|
||
import os
|
||
from dataclasses import dataclass
|
||
from datetime import datetime, timedelta, timezone
|
||
from pathlib import Path
|
||
import posixpath
|
||
import re
|
||
import shutil
|
||
import sys
|
||
import time
|
||
from typing import Protocol
|
||
from urllib.error import HTTPError, URLError
|
||
from urllib.parse import quote, urlencode
|
||
from urllib.request import Request, urlopen
|
||
import warnings
|
||
import zipfile
|
||
|
||
|
||
VENDOR_DIR = Path(__file__).resolve().parent / "vendor"
|
||
sys.path.insert(0, str(VENDOR_DIR))
|
||
|
||
from openpyxl import load_workbook
|
||
|
||
try:
|
||
from runtime_config import METRIX_API_BASE_URL, METRIX_API_TOKEN, METRIX_DATABASE_CONNECTION_ID
|
||
except ImportError:
|
||
METRIX_API_BASE_URL = "http://188.5.127.115:18271"
|
||
METRIX_API_TOKEN = ""
|
||
METRIX_DATABASE_CONNECTION_ID = ""
|
||
|
||
|
||
EXPECTED_TYPES = (
|
||
"5G下FDD干扰监控",
|
||
"5G干扰监控",
|
||
"700M下FDD干扰监控",
|
||
"700M干扰监控",
|
||
"SDR_FDD干扰监控",
|
||
"SDR_TDD干扰监控",
|
||
"反开RD干扰监控",
|
||
)
|
||
|
||
DEFAULT_STORAGE_ID = "stg_4d9a910d72"
|
||
DEFAULT_ROOT = "/网优日常优化数据文档/(勿删)干扰定时小时指标"
|
||
CELL_DATA_DIRECTORIES = (
|
||
"/网优日常优化数据文档/日常性能报表/2026年/5G/5G小区信息表",
|
||
"/网优日常优化数据文档/日常性能报表/2026年/700M/700M小区信息表",
|
||
"/网优日常优化数据文档/日常性能报表/2026年/5G_反开/反开小区信息表",
|
||
"/网优日常优化数据文档/日常性能报表/2026年/TDD_LTE/LTE基础信息数据/LTE_小区信息表",
|
||
"/网优日常优化数据文档/日常性能报表/2026年/FDD_LTE/FDD基础信息数据/FDD小区信息表",
|
||
)
|
||
FILE_RE = re.compile(r"^(?P<source_type>.+)_LWP_每小时_过滤110_(?P<window>\d{16})\.zip$", re.IGNORECASE)
|
||
CELL_DATA_FILE_RE = re.compile(r"(?P<date>20\d{6})\.xlsx$", re.IGNORECASE)
|
||
|
||
LTE_PLMN = "460-00"
|
||
DATABASE_NAME = "interference_etl"
|
||
DATABASE_TABLE = "interference_hourly_summary"
|
||
|
||
HEADER_FDD = (
|
||
"开始时间",
|
||
"粒度",
|
||
"子网ID",
|
||
"子网名称",
|
||
"网元ID",
|
||
"管理网元",
|
||
"eNodeB CUID",
|
||
"eNodeB CU名称",
|
||
"LTEID",
|
||
"LTE名称",
|
||
"E-UTRAN FDD小区ID",
|
||
"E-UTRAN FDD小区名称",
|
||
"cellId",
|
||
"eNodeBId",
|
||
"载波平均噪声干扰(dBm)",
|
||
"集团-上下行总业务量(GB)",
|
||
"RRC连接建立最大用户数",
|
||
)
|
||
|
||
HEADER_NR = (
|
||
"开始时间",
|
||
"粒度",
|
||
"子网ID",
|
||
"子网名称",
|
||
"网元ID",
|
||
"管理网元",
|
||
"gNB CU-CP功能配置ID",
|
||
"gNB CU-CP功能配置名称",
|
||
"CU小区配置ID",
|
||
"CU小区配置名称",
|
||
"cellId",
|
||
"duMeMoId",
|
||
"gNBId",
|
||
"gNBIdLength",
|
||
"gNBplmn",
|
||
"masterOperatorId",
|
||
"nrCarrierGroupId",
|
||
"nrPhysicalCellDUId",
|
||
"小区上行平均干扰电平(dBm)",
|
||
"5G上下行总流量(上行PDCP PDU数据量+下行PDCP成功发送数据量)(GB)",
|
||
"RRC连接最大连接用户数",
|
||
)
|
||
|
||
HEADER_SDR = (
|
||
"开始时间",
|
||
"粒度",
|
||
"子网ID",
|
||
"子网名称",
|
||
"网元ID",
|
||
"管理网元",
|
||
"eNodeBID",
|
||
"eNodeB名称",
|
||
"小区ID",
|
||
"小区名称",
|
||
"载波平均噪声干扰(dBm)",
|
||
"集团-上下行总业务量(GB)",
|
||
"RRC连接建立最大用户数",
|
||
)
|
||
|
||
HEADER_RD = (
|
||
"开始时间",
|
||
"粒度",
|
||
"子网ID",
|
||
"子网名称",
|
||
"网元ID",
|
||
"管理网元",
|
||
"eNodeB CUID",
|
||
"eNodeB CU名称",
|
||
"LTEID",
|
||
"LTE名称",
|
||
"E-UTRAN TDD小区ID",
|
||
"E-UTRAN TDD小区名称",
|
||
"cellId",
|
||
"eNodeBId",
|
||
"载波平均噪声干扰(dBm)",
|
||
"上下行总业务量(GB)",
|
||
"RRC连接建立最大用户数",
|
||
)
|
||
|
||
EXPECTED_HEADERS = {
|
||
"5G下FDD干扰监控": HEADER_FDD,
|
||
"5G干扰监控": HEADER_NR,
|
||
"700M下FDD干扰监控": HEADER_FDD,
|
||
"700M干扰监控": HEADER_NR,
|
||
"SDR_FDD干扰监控": HEADER_SDR,
|
||
"SDR_TDD干扰监控": HEADER_SDR,
|
||
"反开RD干扰监控": HEADER_RD,
|
||
}
|
||
|
||
SUMMARY_HEADER = (
|
||
"hour_start",
|
||
"hour_end",
|
||
"cgi",
|
||
"cell_name",
|
||
"interference_dbm",
|
||
"longitude",
|
||
"latitude",
|
||
)
|
||
|
||
class ProcessingError(RuntimeError):
|
||
pass
|
||
|
||
|
||
@dataclass(frozen=True)
|
||
class Candidate:
|
||
source_type: str
|
||
window: str
|
||
path: str
|
||
size: int = 0
|
||
|
||
|
||
class Source(Protocol):
|
||
def candidates(self, lookback_days: int) -> list[Candidate]: ...
|
||
|
||
def download(self, path: str) -> bytes: ...
|
||
|
||
def cell_data_workbooks(self) -> list[tuple[str, bytes]]: ...
|
||
|
||
|
||
class SummaryStore(Protocol):
|
||
def replace_latest(self, rows: list[dict[str, str]]) -> dict[str, object]: ...
|
||
|
||
|
||
class MetrixApiClient:
|
||
def __init__(self, base_url: str, token: str) -> None:
|
||
if not token:
|
||
raise ProcessingError("METRIX_API_TOKEN is required for Metrix API access")
|
||
self.base_url = base_url.rstrip("/")
|
||
self.token = token
|
||
|
||
def get_bytes(self, endpoint: str, query: dict[str, object] | None = None, timeout: int = 30) -> bytes:
|
||
return self._request("GET", endpoint, query=query, timeout=timeout)
|
||
|
||
def get_json(self, endpoint: str, query: dict[str, object] | None = None, timeout: int = 30) -> object:
|
||
return self._decode_json(self.get_bytes(endpoint, query, timeout))
|
||
|
||
def post_json(self, endpoint: str, payload: dict[str, object], timeout: int = 30) -> object:
|
||
body = json.dumps(payload, ensure_ascii=False).encode("utf-8")
|
||
return self._decode_json(self._request("POST", endpoint, body=body, timeout=timeout))
|
||
|
||
def _request(
|
||
self,
|
||
method: str,
|
||
endpoint: str,
|
||
query: dict[str, object] | None = None,
|
||
body: bytes | None = None,
|
||
timeout: int = 30,
|
||
) -> bytes:
|
||
url = f"{self.base_url}{endpoint}"
|
||
if query:
|
||
url = f"{url}?{urlencode(query)}"
|
||
headers = {"Authorization": f"Bearer {self.token}", "Accept": "application/json"}
|
||
if body is not None:
|
||
headers["Content-Type"] = "application/json"
|
||
request = Request(url, data=body, headers=headers, method=method)
|
||
last_error: Exception | None = None
|
||
for attempt in range(3):
|
||
try:
|
||
with urlopen(request, timeout=timeout) as response:
|
||
return response.read()
|
||
except HTTPError as exc:
|
||
detail = exc.read().decode("utf-8", "replace")[:1000]
|
||
if exc.code < 500:
|
||
raise ProcessingError(f"Metrix API {exc.code}: {detail}") from exc
|
||
last_error = exc
|
||
except (URLError, TimeoutError) as exc:
|
||
last_error = exc
|
||
if attempt < 2:
|
||
time.sleep(2**attempt)
|
||
raise ProcessingError(f"Metrix API request failed: {last_error}")
|
||
|
||
@staticmethod
|
||
def _decode_json(payload: bytes) -> object:
|
||
try:
|
||
return json.loads(payload.decode("utf-8"))
|
||
except (UnicodeDecodeError, json.JSONDecodeError) as exc:
|
||
raise ProcessingError("Metrix API returned invalid JSON") from exc
|
||
|
||
|
||
class ApiSummaryStore:
|
||
def __init__(self, client: MetrixApiClient, connection_id: str) -> None:
|
||
if not connection_id:
|
||
raise ProcessingError("METRIX_DATABASE_CONNECTION_ID is required for database output")
|
||
self.client = client
|
||
self.connection_id = connection_id
|
||
self.table = DATABASE_TABLE
|
||
|
||
def replace_latest(self, rows: list[dict[str, str]]) -> dict[str, object]:
|
||
if not rows:
|
||
raise ProcessingError("Refusing to replace database data with an empty batch")
|
||
metric_times = {datetime.strptime(row["hour_start"], "%Y-%m-%d %H:%M:%S") for row in rows}
|
||
if len(metric_times) != 1:
|
||
raise ProcessingError("Database batch must contain exactly one metric hour")
|
||
metric_time = next(iter(metric_times))
|
||
self._query(
|
||
f"CREATE DATABASE IF NOT EXISTS `{DATABASE_NAME}` "
|
||
"CHARACTER SET utf8mb4 COLLATE utf8mb4_unicode_ci"
|
||
)
|
||
self._query(
|
||
f"""
|
||
CREATE TABLE IF NOT EXISTS `{self.table}` (
|
||
metric_time DATETIME NOT NULL COMMENT '指标开始时间',
|
||
cgi VARCHAR(128) NOT NULL,
|
||
cell_name VARCHAR(255) NOT NULL,
|
||
interference_dbm DECIMAL(10,3) NOT NULL,
|
||
longitude DECIMAL(10,6) NULL,
|
||
latitude DECIMAL(10,6) NULL,
|
||
PRIMARY KEY (metric_time, cgi)
|
||
) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4
|
||
""",
|
||
database=DATABASE_NAME,
|
||
)
|
||
columns = self.client.get_json(
|
||
self._endpoint("columns"),
|
||
{"database": DATABASE_NAME, "table": self.table},
|
||
)
|
||
if not isinstance(columns, list):
|
||
raise ProcessingError("Metrix Database API returned invalid column metadata")
|
||
existing_columns = {str(item.get("name")) for item in columns if isinstance(item, dict)}
|
||
for column in ("longitude", "latitude"):
|
||
if column not in existing_columns:
|
||
self._query(
|
||
f"ALTER TABLE `{self.table}` ADD COLUMN `{column}` DECIMAL(10,6) NULL",
|
||
database=DATABASE_NAME,
|
||
)
|
||
|
||
latest_payload = self._query(
|
||
f"SELECT MAX(metric_time) AS latest_time FROM `{self.table}`",
|
||
database=DATABASE_NAME,
|
||
)
|
||
latest_rows = latest_payload.get("rows", [])
|
||
latest_value = latest_rows[0].get("latest_time") if latest_rows else None
|
||
latest_time = datetime.fromisoformat(str(latest_value)) if latest_value else None
|
||
if latest_time is not None and latest_time > metric_time:
|
||
raise ProcessingError(
|
||
f"Database already contains newer metric time {latest_time:%Y-%m-%d %H:%M:%S}; "
|
||
f"refusing to replace it with {metric_time:%Y-%m-%d %H:%M:%S}"
|
||
)
|
||
|
||
metric_literal = f"'{metric_time:%Y-%m-%d %H:%M:%S}'"
|
||
values = ",\n".join(self._row_values(metric_literal, row) for row in rows)
|
||
script_payload = self.client.post_json(
|
||
self._endpoint("run-script"),
|
||
{
|
||
"content": f"""
|
||
START TRANSACTION;
|
||
DELETE FROM `{self.table}` WHERE metric_time = {metric_literal};
|
||
INSERT INTO `{self.table}`
|
||
(metric_time, cgi, cell_name, interference_dbm, longitude, latitude)
|
||
VALUES
|
||
{values};
|
||
DELETE FROM `{self.table}` WHERE metric_time <> {metric_literal};
|
||
COMMIT;
|
||
""",
|
||
"database": DATABASE_NAME,
|
||
"stop_on_error": True,
|
||
"single_session": True,
|
||
},
|
||
timeout=120,
|
||
)
|
||
if not isinstance(script_payload, dict) or not isinstance(script_payload.get("results"), list):
|
||
raise ProcessingError("Metrix Database API returned an invalid script result")
|
||
results = script_payload["results"]
|
||
failed = next((item for item in results if not item.get("ok")), None)
|
||
if failed is not None:
|
||
raise ProcessingError(f"Database script failed at statement {failed.get('index')}: {failed.get('message', '')}")
|
||
if len(results) != 5:
|
||
raise ProcessingError(f"Database script returned {len(results)} results; expected 5")
|
||
|
||
return {
|
||
"enabled": True,
|
||
"table": self.table,
|
||
"metric_time": metric_time.strftime("%Y-%m-%d %H:%M:%S"),
|
||
"inserted_rows": int(results[2].get("affected_rows") or 0),
|
||
"refreshed_rows": int(results[1].get("affected_rows") or 0),
|
||
"old_rows_deleted": int(results[3].get("affected_rows") or 0),
|
||
}
|
||
|
||
def _endpoint(self, action: str) -> str:
|
||
return f"/api/databases/{quote(self.connection_id, safe='')}/{action}"
|
||
|
||
def _query(self, sql: str, database: str = "") -> dict[str, object]:
|
||
payload = self.client.post_json(
|
||
self._endpoint("query"),
|
||
{"sql": sql, "database": database, "page": 1, "page_size": 100},
|
||
timeout=120,
|
||
)
|
||
if not isinstance(payload, dict):
|
||
raise ProcessingError("Metrix Database API returned an invalid query result")
|
||
return payload
|
||
|
||
@staticmethod
|
||
def _row_values(metric_literal: str, row: dict[str, str]) -> str:
|
||
return "(" + ", ".join(
|
||
(
|
||
metric_literal,
|
||
sql_text_literal(row["cgi"]),
|
||
sql_text_literal(row["cell_name"]),
|
||
sql_decimal_literal(row["interference_dbm"], "interference_dbm"),
|
||
sql_decimal_literal(row["longitude"], "longitude", nullable=True),
|
||
sql_decimal_literal(row["latitude"], "latitude", nullable=True),
|
||
)
|
||
) + ")"
|
||
|
||
|
||
class ApiSource:
|
||
def __init__(self, client: MetrixApiClient, storage_id: str, root: str) -> None:
|
||
self.client = client
|
||
self.storage_id = storage_id
|
||
self.root = root.rstrip("/") or "/"
|
||
|
||
def candidates(self, lookback_days: int) -> list[Candidate]:
|
||
root_entries = self._list_dir(self.root)
|
||
date_dirs = sorted(
|
||
(item for item in root_entries if item.get("is_dir") and re.fullmatch(r"\d{4}-\d{2}-\d{2}", item.get("name", ""))),
|
||
key=lambda item: item["name"],
|
||
)
|
||
selected_dirs = date_dirs[-max(1, lookback_days) :]
|
||
result: list[Candidate] = []
|
||
for directory in selected_dirs:
|
||
for item in self._list_dir(directory["path"]):
|
||
if item.get("is_dir"):
|
||
continue
|
||
candidate = parse_candidate(item["path"], int(item.get("size") or 0))
|
||
if candidate is not None:
|
||
result.append(candidate)
|
||
return result
|
||
|
||
def download(self, path: str) -> bytes:
|
||
endpoint = f"/api/storages/{quote(self.storage_id, safe='')}/download"
|
||
return self.client.get_bytes(endpoint, {"path": path}, timeout=120)
|
||
|
||
def cell_data_workbooks(self) -> list[tuple[str, bytes]]:
|
||
workbooks: list[tuple[str, bytes]] = []
|
||
for directory in CELL_DATA_DIRECTORIES:
|
||
item = select_latest_cell_data_file(self._list_dir(directory), directory)
|
||
path = str(item["path"])
|
||
workbooks.append((path, self.download(path)))
|
||
return workbooks
|
||
|
||
def _list_dir(self, path: str) -> list[dict[str, object]]:
|
||
endpoint = f"/api/storages/{quote(self.storage_id, safe='')}/files"
|
||
payload = self.client.get_json(endpoint, {"path": path, "recursive": "false"})
|
||
if not isinstance(payload, dict) or not isinstance(payload.get("entries"), list):
|
||
raise ProcessingError("Metrix Storage API returned an invalid file list")
|
||
return payload["entries"]
|
||
|
||
|
||
class LocalSource:
|
||
def __init__(self, root: Path) -> None:
|
||
self.root = root.resolve()
|
||
if not self.root.is_dir():
|
||
raise ProcessingError(f"Local source directory does not exist: {self.root}")
|
||
|
||
def candidates(self, lookback_days: int) -> list[Candidate]:
|
||
del lookback_days
|
||
result: list[Candidate] = []
|
||
for path in self.root.rglob("*.zip"):
|
||
candidate = parse_candidate(str(path.resolve()), path.stat().st_size)
|
||
if candidate is not None:
|
||
result.append(candidate)
|
||
return result
|
||
|
||
def download(self, path: str) -> bytes:
|
||
return Path(path).read_bytes()
|
||
|
||
def cell_data_workbooks(self) -> list[tuple[str, bytes]]:
|
||
directories = [self.root / Path(directory.lstrip("/")) for directory in CELL_DATA_DIRECTORIES]
|
||
existing = [directory for directory in directories if directory.is_dir()]
|
||
if not existing:
|
||
return []
|
||
if len(existing) != len(directories):
|
||
raise ProcessingError("Local CellData source must contain all five configured directories")
|
||
workbooks: list[tuple[str, bytes]] = []
|
||
for directory in directories:
|
||
entries = [
|
||
{"name": path.name, "path": str(path.resolve()), "is_dir": path.is_dir()}
|
||
for path in directory.iterdir()
|
||
]
|
||
item = select_latest_cell_data_file(entries, str(directory))
|
||
path = str(item["path"])
|
||
workbooks.append((path, self.download(path)))
|
||
return workbooks
|
||
|
||
|
||
def parse_candidate(path: str, size: int = 0) -> Candidate | None:
|
||
name = posixpath.basename(path.replace("\\", "/"))
|
||
match = FILE_RE.fullmatch(name)
|
||
if not match or match.group("source_type") not in EXPECTED_TYPES:
|
||
return None
|
||
return Candidate(match.group("source_type"), match.group("window"), path, size)
|
||
|
||
|
||
def select_latest_cell_data_file(entries: list[dict[str, object]], directory: str) -> dict[str, object]:
|
||
dated_files: list[tuple[datetime, str, dict[str, object]]] = []
|
||
for item in entries:
|
||
if item.get("is_dir"):
|
||
continue
|
||
name = str(item.get("name") or "")
|
||
match = CELL_DATA_FILE_RE.search(name)
|
||
if not match:
|
||
continue
|
||
try:
|
||
file_date = datetime.strptime(match.group("date"), "%Y%m%d")
|
||
except ValueError:
|
||
continue
|
||
dated_files.append((file_date, name, item))
|
||
if not dated_files:
|
||
raise ProcessingError(f"No dated CellData XLSX found in {directory}")
|
||
return max(dated_files, key=lambda entry: (entry[0], entry[1]))[2]
|
||
|
||
|
||
def parse_cell_data_workbook(raw_xlsx: bytes, path: str) -> dict[str, tuple[str, str]]:
|
||
try:
|
||
with warnings.catch_warnings():
|
||
warnings.filterwarnings("ignore", message="Workbook contains no default style")
|
||
workbook = load_workbook(io.BytesIO(raw_xlsx), read_only=True, data_only=True)
|
||
except Exception as exc:
|
||
raise ProcessingError(f"Invalid CellData workbook {path}: {exc}") from exc
|
||
try:
|
||
if "小区信息表" not in workbook.sheetnames:
|
||
raise ProcessingError(f"CellData workbook does not contain 小区信息表: {path}")
|
||
rows = workbook["小区信息表"].iter_rows(values_only=True)
|
||
try:
|
||
header = tuple(normalize_cell(value) for value in next(rows))
|
||
except StopIteration as exc:
|
||
raise ProcessingError(f"CellData workbook is empty: {path}") from exc
|
||
required_columns = ("eNB/gNB", "CI", "经度", "纬度")
|
||
missing = [column for column in required_columns if column not in header]
|
||
if missing:
|
||
raise ProcessingError(f"CellData workbook missing columns {missing}: {path}")
|
||
indexes = {column: header.index(column) for column in required_columns}
|
||
coordinates: dict[str, tuple[str, str]] = {}
|
||
for row_number, row in enumerate(rows, start=2):
|
||
node = normalize_cell(row[indexes["eNB/gNB"]])
|
||
cell = normalize_cell(row[indexes["CI"]])
|
||
longitude = normalize_cell(row[indexes["经度"]])
|
||
latitude = normalize_cell(row[indexes["纬度"]])
|
||
if not node or not cell or not longitude or not latitude:
|
||
continue
|
||
cgi = f"{LTE_PLMN}-{node}-{cell}"
|
||
value = (longitude, latitude)
|
||
existing = coordinates.get(cgi)
|
||
if existing is not None and existing != value:
|
||
raise ProcessingError(f"Conflicting CellData coordinates for {cgi} in {path}, row {row_number}")
|
||
coordinates[cgi] = value
|
||
return coordinates
|
||
finally:
|
||
workbook.close()
|
||
|
||
|
||
def load_cell_coordinates(workbooks: list[tuple[str, bytes]]) -> dict[str, tuple[str, str]]:
|
||
coordinates: dict[str, tuple[str, str]] = {}
|
||
for path, raw_xlsx in workbooks:
|
||
for cgi, value in parse_cell_data_workbook(raw_xlsx, path).items():
|
||
existing = coordinates.get(cgi)
|
||
if existing is not None and existing != value:
|
||
raise ProcessingError(f"Conflicting CellData coordinates for {cgi} across workbooks")
|
||
coordinates[cgi] = value
|
||
return coordinates
|
||
|
||
|
||
def select_window(candidates: list[Candidate], requested: str = "") -> tuple[str, dict[str, Candidate], list[str]]:
|
||
grouped: dict[str, dict[str, Candidate]] = {}
|
||
duplicates: list[str] = []
|
||
for candidate in candidates:
|
||
window_group = grouped.setdefault(candidate.window, {})
|
||
if candidate.source_type in window_group:
|
||
duplicates.append(f"{candidate.window}/{candidate.source_type}")
|
||
window_group[candidate.source_type] = candidate
|
||
if duplicates:
|
||
raise ProcessingError(f"Duplicate source files: {', '.join(sorted(duplicates))}")
|
||
complete = sorted(window for window, items in grouped.items() if all(name in items for name in EXPECTED_TYPES))
|
||
if requested:
|
||
if requested not in complete:
|
||
present = sorted(grouped.get(requested, {}))
|
||
missing = [name for name in EXPECTED_TYPES if name not in present]
|
||
raise ProcessingError(f"Requested window is incomplete: {requested}; missing={missing}")
|
||
selected = requested
|
||
elif complete:
|
||
selected = complete[-1]
|
||
else:
|
||
raise ProcessingError("No hour contains all seven interference source types")
|
||
warnings_out: list[str] = []
|
||
latest_seen = max(grouped) if grouped else ""
|
||
if latest_seen and latest_seen != selected:
|
||
missing = [name for name in EXPECTED_TYPES if name not in grouped[latest_seen]]
|
||
warnings_out.append(f"Latest observed window {latest_seen} is incomplete; using {selected}; missing={missing}")
|
||
return selected, grouped[selected], warnings_out
|
||
|
||
|
||
def parse_workbook(raw_zip: bytes, source_type: str) -> tuple[tuple[str, ...], list[tuple[object, ...]], str]:
|
||
try:
|
||
with zipfile.ZipFile(io.BytesIO(raw_zip)) as archive:
|
||
bad_member = archive.testzip()
|
||
if bad_member:
|
||
raise ProcessingError(f"ZIP CRC check failed: {bad_member}")
|
||
xlsx_members = [item for item in archive.infolist() if not item.is_dir() and item.filename.lower().endswith(".xlsx")]
|
||
if len(xlsx_members) != 1:
|
||
raise ProcessingError(f"Expected one XLSX member, found {len(xlsx_members)}")
|
||
member = xlsx_members[0]
|
||
workbook_bytes = archive.read(member)
|
||
except zipfile.BadZipFile as exc:
|
||
raise ProcessingError("Invalid ZIP archive") from exc
|
||
|
||
with warnings.catch_warnings():
|
||
warnings.filterwarnings("ignore", message="Workbook contains no default style")
|
||
workbook = load_workbook(io.BytesIO(workbook_bytes), read_only=True, data_only=True)
|
||
try:
|
||
if "Sheet0" not in workbook.sheetnames:
|
||
raise ProcessingError("Workbook does not contain Sheet0")
|
||
rows = workbook["Sheet0"].iter_rows(values_only=True)
|
||
try:
|
||
header = tuple(normalize_cell(value) for value in next(rows))
|
||
except StopIteration as exc:
|
||
raise ProcessingError("Sheet0 is empty") from exc
|
||
expected = EXPECTED_HEADERS[source_type]
|
||
if header != expected:
|
||
raise ProcessingError(f"Unexpected Sheet0 header for {source_type}: {header}")
|
||
data = [tuple(row) for row in rows if any(value is not None and normalize_cell(value) != "" for value in row)]
|
||
return header, data, member.filename
|
||
finally:
|
||
workbook.close()
|
||
|
||
|
||
def process(
|
||
source: Source,
|
||
output_root: Path,
|
||
lookback_days: int,
|
||
requested_window: str = "",
|
||
store: SummaryStore | None = None,
|
||
) -> Path:
|
||
candidates = source.candidates(lookback_days)
|
||
window, selected, warnings_out = select_window(candidates, requested_window)
|
||
cell_data_workbooks = source.cell_data_workbooks()
|
||
cell_coordinates = load_cell_coordinates(cell_data_workbooks)
|
||
temp_dir = output_root.resolve() / f".{window}.tmp-{os.getpid()}"
|
||
final_dir = output_root.resolve() / window
|
||
ensure_scoped(output_root.resolve(), temp_dir)
|
||
if temp_dir.exists():
|
||
shutil.rmtree(temp_dir)
|
||
converted_dir = temp_dir / "converted"
|
||
converted_dir.mkdir(parents=True)
|
||
|
||
summary_rows: list[dict[str, str]] = []
|
||
manifest_files: list[dict[str, object]] = []
|
||
database_result: dict[str, object] = {"enabled": False}
|
||
try:
|
||
for source_type in EXPECTED_TYPES:
|
||
candidate = selected[source_type]
|
||
raw_zip = source.download(candidate.path)
|
||
header, rows, member_name = parse_workbook(raw_zip, source_type)
|
||
csv_name = f"{source_type}_{window}.csv"
|
||
write_csv(converted_dir / csv_name, header, rows)
|
||
for row_number, row in enumerate(rows, start=2):
|
||
record = dict(zip(header, row, strict=True))
|
||
summary_rows.append(
|
||
summary_record(record, source_type, candidate.path, window, row_number, cell_coordinates)
|
||
)
|
||
manifest_files.append(
|
||
{
|
||
"source_size": candidate.size,
|
||
"sha256": hashlib.sha256(raw_zip).hexdigest(),
|
||
"xlsx_member": member_name,
|
||
"rows": len(rows),
|
||
"converted_csv": f"converted/{csv_name}",
|
||
}
|
||
)
|
||
|
||
summary_rows.sort(key=lambda item: (item["cgi"], item["cell_name"]))
|
||
summary_name = f"interference_summary_{window}.csv"
|
||
write_dict_csv(temp_dir / summary_name, SUMMARY_HEADER, summary_rows)
|
||
if store is not None:
|
||
database_result = store.replace_latest(summary_rows)
|
||
matched_coordinates = sum(bool(row["longitude"] and row["latitude"]) for row in summary_rows)
|
||
manifest = {
|
||
"generated_at": datetime.now(timezone.utc).isoformat(),
|
||
"window": window,
|
||
"hour_start": window_bounds(window)[0],
|
||
"hour_end": window_bounds(window)[1],
|
||
"source_file_count": len(manifest_files),
|
||
"cell_data_file_count": len(cell_data_workbooks),
|
||
"summary_rows": len(summary_rows),
|
||
"coordinate_matched_rows": matched_coordinates,
|
||
"coordinate_unmatched_rows": len(summary_rows) - matched_coordinates,
|
||
"warnings": warnings_out,
|
||
"files": manifest_files,
|
||
"summary_csv": summary_name,
|
||
"database": database_result,
|
||
}
|
||
(temp_dir / "manifest.json").write_text(json.dumps(manifest, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
||
output_root.resolve().mkdir(parents=True, exist_ok=True)
|
||
if final_dir.exists():
|
||
ensure_scoped(output_root.resolve(), final_dir)
|
||
shutil.rmtree(final_dir)
|
||
temp_dir.replace(final_dir)
|
||
except Exception:
|
||
shutil.rmtree(temp_dir, ignore_errors=True)
|
||
raise
|
||
|
||
print(f"selected_window={window}")
|
||
print(f"source_files={len(manifest_files)}")
|
||
print(f"cell_data_files={len(cell_data_workbooks)}")
|
||
print(f"summary_rows={len(summary_rows)}")
|
||
print(f"coordinate_matched_rows={matched_coordinates}")
|
||
print(f"coordinate_unmatched_rows={len(summary_rows) - matched_coordinates}")
|
||
if database_result["enabled"]:
|
||
print(f"database_metric_time={database_result['metric_time']}")
|
||
print(f"database_inserted_rows={database_result['inserted_rows']}")
|
||
print(f"database_old_rows_deleted={database_result['old_rows_deleted']}")
|
||
for message in warnings_out:
|
||
print(f"warning={message}")
|
||
print(f"output={final_dir}")
|
||
return final_dir
|
||
|
||
|
||
def summary_record(
|
||
record: dict[str, object],
|
||
source_type: str,
|
||
source_path: str,
|
||
window: str,
|
||
row_number: int,
|
||
cell_coordinates: dict[str, tuple[str, str]] | None = None,
|
||
) -> dict[str, str]:
|
||
hour_start = normalize_cell(record["开始时间"])
|
||
expected_start, hour_end = window_bounds(window)
|
||
if hour_start != expected_start:
|
||
raise ProcessingError(f"{source_path}: row {row_number} time {hour_start!r} does not match {expected_start!r}")
|
||
|
||
if source_type in ("5G干扰监控", "700M干扰监控"):
|
||
cgi = build_cgi("", record, ("gNBplmn", "gNBId", "cellId"), source_path, row_number)
|
||
cell_name = required(record, "CU小区配置名称", source_path, row_number)
|
||
interference = required(record, "小区上行平均干扰电平(dBm)", source_path, row_number)
|
||
elif source_type in ("SDR_FDD干扰监控", "SDR_TDD干扰监控"):
|
||
cgi = build_cgi(LTE_PLMN, record, ("eNodeBID", "小区ID"), source_path, row_number)
|
||
cell_name = required(record, "小区名称", source_path, row_number)
|
||
interference = required(record, "载波平均噪声干扰(dBm)", source_path, row_number)
|
||
elif source_type == "反开RD干扰监控":
|
||
cgi = build_cgi(LTE_PLMN, record, ("eNodeBId", "cellId"), source_path, row_number)
|
||
cell_name = required(record, "E-UTRAN TDD小区名称", source_path, row_number)
|
||
interference = required(record, "载波平均噪声干扰(dBm)", source_path, row_number)
|
||
else:
|
||
cgi = build_cgi(LTE_PLMN, record, ("eNodeBId", "cellId"), source_path, row_number)
|
||
cell_name = required(record, "E-UTRAN FDD小区名称", source_path, row_number)
|
||
interference = required(record, "载波平均噪声干扰(dBm)", source_path, row_number)
|
||
|
||
longitude, latitude = (cell_coordinates or {}).get(cgi, ("", ""))
|
||
return {
|
||
"hour_start": hour_start,
|
||
"hour_end": hour_end,
|
||
"cgi": cgi,
|
||
"cell_name": cell_name,
|
||
"interference_dbm": interference,
|
||
"longitude": longitude,
|
||
"latitude": latitude,
|
||
}
|
||
|
||
|
||
def build_cgi(prefix: str, record: dict[str, object], columns: tuple[str, ...], path: str, row: int) -> str:
|
||
parts = [required(record, column, path, row) for column in columns]
|
||
return "-".join(([prefix] if prefix else []) + parts)
|
||
|
||
|
||
def required(record: dict[str, object], column: str, path: str, row: int) -> str:
|
||
value = normalize_cell(record[column])
|
||
if not value:
|
||
raise ProcessingError(f"{path}: row {row} has empty {column}")
|
||
return value
|
||
|
||
|
||
def window_bounds(window: str) -> tuple[str, str]:
|
||
if not re.fullmatch(r"\d{16}", window):
|
||
raise ProcessingError(f"Invalid window: {window}")
|
||
start = datetime.strptime(window[:12], "%Y%m%d%H%M")
|
||
end_hour = int(window[12:14])
|
||
end_minute = int(window[14:16])
|
||
end = start.replace(hour=end_hour, minute=end_minute)
|
||
if end <= start:
|
||
end += timedelta(days=1)
|
||
return start.strftime("%Y-%m-%d %H:%M:%S"), end.strftime("%Y-%m-%d %H:%M:%S")
|
||
|
||
|
||
def normalize_cell(value: object) -> str:
|
||
if value is None:
|
||
return ""
|
||
if isinstance(value, datetime):
|
||
return value.strftime("%Y-%m-%d %H:%M:%S")
|
||
if isinstance(value, float) and value.is_integer():
|
||
return str(int(value))
|
||
return str(value).strip()
|
||
|
||
|
||
def sql_text_literal(value: str) -> str:
|
||
encoded = value.encode("utf-8").hex()
|
||
return "''" if not encoded else f"CONVERT(0x{encoded} USING utf8mb4)"
|
||
|
||
|
||
def sql_decimal_literal(value: str, field: str, nullable: bool = False) -> str:
|
||
normalized = value.strip()
|
||
if not normalized and nullable:
|
||
return "NULL"
|
||
try:
|
||
number = Decimal(normalized)
|
||
except InvalidOperation as exc:
|
||
raise ProcessingError(f"Invalid decimal value for {field}: {value}") from exc
|
||
if not number.is_finite():
|
||
raise ProcessingError(f"Invalid decimal value for {field}: {value}")
|
||
return format(number, "f")
|
||
|
||
|
||
def write_csv(path: Path, header: tuple[str, ...], rows: list[tuple[object, ...]]) -> None:
|
||
with path.open("w", encoding="utf-8-sig", newline="") as file:
|
||
writer = csv.writer(file)
|
||
writer.writerow(header)
|
||
for row in rows:
|
||
writer.writerow(normalize_cell(value) for value in row)
|
||
|
||
|
||
def write_dict_csv(path: Path, header: tuple[str, ...], rows: list[dict[str, str]]) -> None:
|
||
with path.open("w", encoding="utf-8-sig", newline="") as file:
|
||
writer = csv.DictWriter(file, fieldnames=header, extrasaction="raise")
|
||
writer.writeheader()
|
||
writer.writerows(rows)
|
||
|
||
|
||
def ensure_scoped(root: Path, target: Path) -> None:
|
||
if target == root or root not in target.parents:
|
||
raise ProcessingError(f"Refusing to modify path outside output root: {target}")
|
||
|
||
|
||
def build_parser() -> argparse.ArgumentParser:
|
||
parser = argparse.ArgumentParser(description="Process the latest complete hour of interference KPI files")
|
||
parser.add_argument("--source-dir", type=Path, help="Use a local source tree instead of the Metrix storage API")
|
||
parser.add_argument("--output-dir", type=Path, default=Path(os.getenv("INTERFERENCE_OUTPUT_DIR", "output")))
|
||
parser.add_argument("--window", default=os.getenv("INTERFERENCE_WINDOW", ""), help="Optional exact 16-digit source window")
|
||
parser.add_argument("--lookback-days", type=int, default=int(os.getenv("INTERFERENCE_LOOKBACK_DAYS", "3")))
|
||
parser.add_argument("--storage-id", default=os.getenv("METRIX_STORAGE_ID", DEFAULT_STORAGE_ID))
|
||
parser.add_argument("--root", default=os.getenv("INTERFERENCE_SOURCE_ROOT", DEFAULT_ROOT))
|
||
parser.add_argument("--no-database", action="store_true", help="Generate files without writing the database")
|
||
return parser
|
||
|
||
|
||
def main(argv: list[str] | None = None) -> int:
|
||
args = build_parser().parse_args(argv)
|
||
if args.lookback_days < 1:
|
||
raise ProcessingError("lookback-days must be at least 1")
|
||
client: MetrixApiClient | None = None
|
||
if args.source_dir:
|
||
source: Source = LocalSource(args.source_dir)
|
||
else:
|
||
client = MetrixApiClient(METRIX_API_BASE_URL, METRIX_API_TOKEN)
|
||
source = ApiSource(client, args.storage_id, args.root)
|
||
if args.no_database:
|
||
store = None
|
||
else:
|
||
client = client or MetrixApiClient(METRIX_API_BASE_URL, METRIX_API_TOKEN)
|
||
store = ApiSummaryStore(client, METRIX_DATABASE_CONNECTION_ID)
|
||
process(source, args.output_dir, args.lookback_days, args.window, store=store)
|
||
return 0
|
||
|
||
|
||
if __name__ == "__main__":
|
||
try:
|
||
raise SystemExit(main())
|
||
except ProcessingError as exc:
|
||
print(f"error={exc}", file=sys.stderr)
|
||
raise SystemExit(1) from exc
|