From 3e8c0f335c8afc98e04a08519cad61d407789f7c Mon Sep 17 00:00:00 2001 From: Nixevol Date: Thu, 6 Aug 2026 15:27:13 +0800 Subject: [PATCH] =?UTF-8?q?feat:=20=E7=BB=9F=E8=AE=A1=E4=B8=80=E5=85=AC?= =?UTF-8?q?=E9=87=8C=E5=86=85=E9=AB=98=E5=B9=B2=E6=89=B0=E5=B0=8F=E5=8C=BA?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- README.md | 6 ++-- aidocs/project_context.md | 6 ++++ main.py | 66 ++++++++++++++++++++++++++++++++++++++- tests/test_pipeline.py | 31 +++++++++++++++++- 4 files changed, 105 insertions(+), 4 deletions(-) diff --git a/README.md b/README.md index d08dfcf..554740e 100644 --- a/README.md +++ b/README.md @@ -21,14 +21,16 @@ For every run, the script also reads the latest filename-dated XLSX from each co The summary columns are: ```text -hour_start,hour_end,network_type,cgi,cell_name,interference_dbm,longitude,latitude,azimuth +hour_start,hour_end,network_type,cgi,cell_name,interference_dbm,longitude,latitude,azimuth,nearby_count ``` `network_type` is derived from the seven source types: both `5G...` sources are `2.6G`, both `700M...` sources are `700M`, and SDR/反开 sources are `4G`. The summary and database keep only high-interference rows: `2.6G >= -107 dBm` and `700M/4G >= -110 dBm`. Converted source CSV files remain full source conversions. +`nearby_count` is the number of other high-interference cells from the same selected hour within 1 km, across all network types. The cell itself is excluded, the 1 km boundary is included, and rows without coordinates use `0`. Candidate cells are found through a 1 km spatial grid index and confirmed with exact Haversine distance. + The script never modifies or deletes source storage files. By default it writes the complete selected hour through the Metrix Database API and keeps only that hour in the target table. -The database table contains one time column, `metric_time DATETIME`, which is the source KPI start time, `network_type VARCHAR(16)`, nullable `longitude` and `latitude` columns, and `azimuth DECIMAL(6,2) NOT NULL DEFAULT 0`. A same-hour rerun replaces that whole batch. After a successful insert, rows for every other hour are deleted in the same transaction. The script refuses to replace a newer database hour with an older source hour. +The database table contains one time column, `metric_time DATETIME`, which is the source KPI start time, `network_type VARCHAR(16)`, nullable `longitude` and `latitude` columns, `azimuth DECIMAL(6,2) NOT NULL DEFAULT 0`, and `nearby_count INT NOT NULL DEFAULT 0`. A same-hour rerun replaces that whole batch. After a successful insert, rows for every other hour are deleted in the same transaction. The script refuses to replace a newer database hour with an older source hour. CGI is generated with fixed rules: diff --git a/aidocs/project_context.md b/aidocs/project_context.md index a9701a6..d37ecc7 100644 --- a/aidocs/project_context.md +++ b/aidocs/project_context.md @@ -72,3 +72,9 @@ - The seven converted source CSV files remain unfiltered source conversions. `manifest.json` and stdout record `threshold_filtered_rows` for the summary filter. - Real read-only validation of window `2026080514001500` reduced 1,868 source rows to 1,109 summary rows: `2.6G=137`, `700M=108`, `4G=864`, with 759 lower-interference rows removed. Thirteen containerized unit tests pass. - SSH deployment run `98a44e88b3684bd1b9ad7edcb866fb17` succeeded for the same window. The database contains exactly one hour (`2026-08-05 14:00:00`) and 1,109 rows; grouped API verification found zero threshold violations and confirmed minimums `2.6G=-106.970`, `700M=-110.000`, and `4G=-109.996`. + +## 2026-08-06: Nearby high-interference cell count + +- Summary CSV and `interference_hourly_summary` add `nearby_count`, the number of other retained high-interference cells within 1 km during the same selected hour. Counts include all network types, exclude the row itself, include the exact 1 km boundary, and default to `0` when coordinates are unavailable. +- Candidate lookup uses a dependency-free 1 km Earth-centered three-dimensional grid index. Only the current and 26 adjacent buckets are checked, then Haversine distance confirms the exact radius; this avoids a full all-pairs scan while preserving distance accuracy. +- Unit tests pass in both the project virtual environment and `interference-etl-runtime:1.1` image. A read-only real run for window `2026080613001400` produced 1,116 rows, including 1,110 with coordinates, 925 with non-zero nearby counts, and a maximum count of 39. All 1,116 indexed results matched a separate brute-force comparison, which found 3,973 qualifying pairs. diff --git a/main.py b/main.py index c0e51dd..11c0a0b 100644 --- a/main.py +++ b/main.py @@ -6,6 +6,7 @@ from decimal import Decimal, InvalidOperation import hashlib import io import json +import math import os from dataclasses import dataclass from datetime import datetime, timedelta, timezone @@ -62,6 +63,9 @@ MIN_INTERFERENCE_BY_NETWORK_TYPE = { "4G": Decimal("-110"), } +EARTH_RADIUS_KM = 6371.0088 +NEARBY_RADIUS_KM = 1.0 + DEFAULT_STORAGE_ID = "stg_4d9a910d72" DEFAULT_ROOT = "/网优日常优化数据文档/(勿删)干扰定时小时指标" CELL_DATA_DIRECTORIES = ( @@ -178,6 +182,7 @@ SUMMARY_HEADER = ( "longitude", "latitude", "azimuth", + "nearby_count", ) class ProcessingError(RuntimeError): @@ -290,6 +295,7 @@ class ApiSummaryStore: longitude DECIMAL(10,6) NULL, latitude DECIMAL(10,6) NULL, azimuth DECIMAL(6,2) NOT NULL DEFAULT 0, + nearby_count INT NOT NULL DEFAULT 0, PRIMARY KEY (metric_time, cgi) ) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4 """, @@ -307,6 +313,7 @@ class ApiSummaryStore: "longitude": "DECIMAL(10,6) NULL", "latitude": "DECIMAL(10,6) NULL", "azimuth": "DECIMAL(6,2) NOT NULL DEFAULT 0", + "nearby_count": "INT NOT NULL DEFAULT 0", } for column, definition in migrations.items(): if column not in existing_columns: @@ -337,7 +344,7 @@ class ApiSummaryStore: START TRANSACTION; DELETE FROM `{self.table}` WHERE metric_time = {metric_literal}; INSERT INTO `{self.table}` - (metric_time, network_type, cgi, cell_name, interference_dbm, longitude, latitude, azimuth) + (metric_time, network_type, cgi, cell_name, interference_dbm, longitude, latitude, azimuth, nearby_count) VALUES {values}; DELETE FROM `{self.table}` WHERE metric_time <> {metric_literal}; @@ -392,6 +399,7 @@ class ApiSummaryStore: sql_decimal_literal(row["longitude"], "longitude", nullable=True), sql_decimal_literal(row["latitude"], "latitude", nullable=True), sql_decimal_literal(row["azimuth"], "azimuth"), + str(int(row["nearby_count"])), ) ) + ")" @@ -670,6 +678,7 @@ def process( } ) + populate_nearby_counts(summary_rows) summary_rows.sort(key=lambda item: (item["cgi"], item["cell_name"])) summary_name = f"interference_summary_{window}.csv" write_dict_csv(temp_dir / summary_name, SUMMARY_HEADER, summary_rows) @@ -760,6 +769,7 @@ def summary_record( "longitude": longitude, "latitude": latitude, "azimuth": azimuth, + "nearby_count": "0", } @@ -774,6 +784,60 @@ def passes_interference_threshold(record: dict[str, str]) -> bool: return interference >= MIN_INTERFERENCE_BY_NETWORK_TYPE[network_type] +def populate_nearby_counts(rows: list[dict[str, str]]) -> None: + counts = [0] * len(rows) + spatial_index: dict[tuple[int, int, int], list[tuple[int, float, float]]] = {} + for index, row in enumerate(rows): + row["nearby_count"] = "0" + if not row["longitude"] or not row["latitude"]: + continue + try: + longitude = float(row["longitude"]) + latitude = float(row["latitude"]) + except ValueError as exc: + raise ProcessingError(f"Invalid coordinates for CGI {row['cgi']}") from exc + if not math.isfinite(longitude) or not math.isfinite(latitude) or not -180 <= longitude <= 180 or not -90 <= latitude <= 90: + raise ProcessingError(f"Invalid coordinates for CGI {row['cgi']}") + + bucket = spatial_bucket(longitude, latitude) + for x_offset in (-1, 0, 1): + for y_offset in (-1, 0, 1): + for z_offset in (-1, 0, 1): + nearby_bucket = (bucket[0] + x_offset, bucket[1] + y_offset, bucket[2] + z_offset) + for other_index, other_longitude, other_latitude in spatial_index.get(nearby_bucket, ()): + if haversine_distance_km(longitude, latitude, other_longitude, other_latitude) <= NEARBY_RADIUS_KM: + counts[index] += 1 + counts[other_index] += 1 + spatial_index.setdefault(bucket, []).append((index, longitude, latitude)) + + for index, count in enumerate(counts): + rows[index]["nearby_count"] = str(count) + + +def spatial_bucket(longitude: float, latitude: float) -> tuple[int, int, int]: + longitude = math.radians(longitude) + latitude = math.radians(latitude) + radius_at_latitude = EARTH_RADIUS_KM * math.cos(latitude) + return ( + math.floor(radius_at_latitude * math.cos(longitude) / NEARBY_RADIUS_KM), + math.floor(radius_at_latitude * math.sin(longitude) / NEARBY_RADIUS_KM), + math.floor(EARTH_RADIUS_KM * math.sin(latitude) / NEARBY_RADIUS_KM), + ) + + +def haversine_distance_km(longitude1: float, latitude1: float, longitude2: float, latitude2: float) -> float: + longitude1, latitude1, longitude2, latitude2 = map( + math.radians, + (longitude1, latitude1, longitude2, latitude2), + ) + longitude_delta = longitude2 - longitude1 + latitude_delta = latitude2 - latitude1 + haversine = math.sin(latitude_delta / 2) ** 2 + ( + math.cos(latitude1) * math.cos(latitude2) * math.sin(longitude_delta / 2) ** 2 + ) + return 2 * EARTH_RADIUS_KM * math.asin(math.sqrt(min(1.0, haversine))) + + def build_cgi(prefix: str, record: dict[str, object], columns: tuple[str, ...], path: str, row: int) -> str: parts = [required(record, column, path, row) for column in columns] return "-".join(([prefix] if prefix else []) + parts) diff --git a/tests/test_pipeline.py b/tests/test_pipeline.py index ec8f661..cd3a7f4 100644 --- a/tests/test_pipeline.py +++ b/tests/test_pipeline.py @@ -48,6 +48,7 @@ class PipelineTest(unittest.TestCase): "longitude", "latitude", "azimuth", + "nearby_count", ], ) rows_by_type = {row["cell_name"].removesuffix("-小区"): row for row in rows} @@ -62,8 +63,11 @@ class PipelineTest(unittest.TestCase): self.assertEqual(rows_by_type["5G干扰监控"]["longitude"], "113.123456") self.assertEqual(rows_by_type["5G干扰监控"]["latitude"], "22.654321") self.assertEqual(rows_by_type["5G干扰监控"]["azimuth"], "30") + self.assertEqual(rows_by_type["5G干扰监控"]["nearby_count"], "1") + self.assertEqual(rows_by_type["700M干扰监控"]["nearby_count"], "1") self.assertTrue(all(not row["longitude"] and not row["latitude"] for row in rows if row["cgi"] == "460-00-100-1")) self.assertTrue(all(row["azimuth"] == "0" for row in rows if row["cgi"] == "460-00-100-1")) + self.assertTrue(all(row["nearby_count"] == "0" for row in rows if row["cgi"] == "460-00-100-1")) manifest = json.loads((result / "manifest.json").read_text(encoding="utf-8")) self.assertEqual(manifest["source_file_count"], 7) @@ -153,6 +157,21 @@ class PipelineTest(unittest.TestCase): self.assertEqual(metadata["460-00-200-1"], ("113.123456", "22.654321", "0")) + def test_nearby_count_uses_high_interference_rows_with_coordinates(self) -> None: + rows = [ + nearby_row("a", "113.000000", "22.000000"), + nearby_row("b", "113.005000", "22.000000"), + nearby_row("c", "113.020000", "22.000000"), + nearby_row("d", "113.000000", "22.000000"), + nearby_row("e", "", ""), + ] + + main.populate_nearby_counts(rows) + + self.assertEqual([row["nearby_count"] for row in rows], ["2", "2", "0", "2", "0"]) + self.assertLess(main.haversine_distance_km(113.0, 22.0, 113.005, 22.0), 1.0) + self.assertGreater(main.haversine_distance_km(113.0, 22.0, 113.02, 22.0), 1.0) + def test_api_store_replaces_same_hour_and_deletes_other_hours(self) -> None: client = FakeApiClient("2026-07-31T10:00:00") store = main.ApiSummaryStore(client, "db_share_mysql") @@ -169,10 +188,11 @@ class PipelineTest(unittest.TestCase): self.assertTrue(any("ADD COLUMN `longitude`" in query for query in queries)) self.assertTrue(any("ADD COLUMN `latitude`" in query for query in queries)) self.assertTrue(any("ADD COLUMN `azimuth`" in query for query in queries)) + self.assertTrue(any("ADD COLUMN `nearby_count`" in query for query in queries)) script = next(payload for endpoint, payload in client.posts if endpoint.endswith("/run-script")) self.assertTrue(script["single_session"]) self.assertIn("START TRANSACTION", script["content"]) - self.assertIn("NULL, NULL, 0", script["content"]) + self.assertIn("NULL, NULL, 0, 0", script["content"]) self.assertIn("CONVERT(0x", script["content"]) self.assertNotIn("source_type", script["content"]) self.assertNotIn("source_path", script["content"]) @@ -273,9 +293,18 @@ def database_row() -> dict[str, str]: "longitude": "", "latitude": "", "azimuth": "0", + "nearby_count": "0", } +def nearby_row(cgi: str, longitude: str, latitude: str) -> dict[str, str]: + row = database_row() + row["cgi"] = cgi + row["longitude"] = longitude + row["latitude"] = latitude + return row + + def create_archive( root: Path, source_type: str,