From 5e4b58255f4019be1e03bffc967cc28b5acad2ab Mon Sep 17 00:00:00 2001 From: Nixevol Date: Fri, 7 Aug 2026 09:22:26 +0800 Subject: [PATCH] =?UTF-8?q?feat:=20=E5=A2=9E=E5=8A=A0=E4=B8=8A=E4=B8=80?= =?UTF-8?q?=E5=91=A8=E6=9C=9F=E5=B9=B2=E6=89=B0=E5=AF=B9=E6=AF=94=E5=AD=97?= =?UTF-8?q?=E6=AE=B5?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- README.md | 6 ++++-- aidocs/project_context.md | 6 ++++++ main.py | 32 ++++++++++++++++++++++++++++++-- tests/test_pipeline.py | 26 +++++++++++++++++++++++++- 4 files changed, 65 insertions(+), 5 deletions(-) diff --git a/README.md b/README.md index 7ec0908..ac37d20 100644 --- a/README.md +++ b/README.md @@ -25,7 +25,7 @@ For every processed group, the script also reads the latest filename-dated XLSX The summary columns are: ```text -metric_time,network_type,cgi,cell_name,interference_dbm,longitude,latitude,azimuth,nearby_count +metric_time,network_type,cgi,cell_name,interference_dbm,prev_interference_dbm,longitude,latitude,azimuth,nearby_count,prev_nearby_count ``` All selected workbook rows must belong to the selected natural 15-minute group. Every summary row receives the same normalized `metric_time`; a row outside the group fails the run before database or history output. @@ -34,9 +34,11 @@ All selected workbook rows must belong to the selected natural 15-minute group. `nearby_count` is the number of other high-interference cells from the same selected time group within 1 km, across all network types. The cell itself is excluded, the 1 km boundary is included, and rows without coordinates use `0`. Candidate cells are found through a 1 km spatial grid index and confirmed with exact Haversine distance. +`prev_interference_dbm` and `prev_nearby_count` come from the previously retained database time, matched by CGI before the current batch replaces old rows. They represent the same cell's previous-period interference value and nearby high-interference count. On the first run, or when a CGI did not exist in the previous period, both fields are empty in CSV/history output and `NULL` in MySQL. + The script never modifies or deletes source storage files. Before downloading source ZIP files or CellData, it compares the normalized target time with the database. A target already present, or older than the database, is not processed again. -The database table contains normalized `metric_time DATETIME`, `network_type VARCHAR(16)`, nullable `longitude` and `latitude`, `azimuth DECIMAL(6,2) NOT NULL DEFAULT 0`, and `nearby_count INT NOT NULL DEFAULT 0`. A successful transaction replaces the target batch and deletes every other database time, so the table retains only the latest processed group. +The database table contains normalized `metric_time DATETIME`, `network_type VARCHAR(16)`, `interference_dbm DECIMAL(10,3)`, nullable `prev_interference_dbm DECIMAL(10,3)`, nullable `longitude` and `latitude`, `azimuth DECIMAL(6,2) NOT NULL DEFAULT 0`, `nearby_count INT NOT NULL DEFAULT 0`, and nullable `prev_nearby_count INT`. A successful transaction replaces the target batch and deletes every other database time, so the table retains only the latest processed group. The previous-period values are read before this replacement transaction. After the database succeeds, the same nine result columns are uploaded as UTF-8-BOM CSV to: diff --git a/aidocs/project_context.md b/aidocs/project_context.md index 3bbbd18..462f9a2 100644 --- a/aidocs/project_context.md +++ b/aidocs/project_context.md @@ -91,3 +91,9 @@ - Database verification found only `2026-08-06 15:00:00` and 1,082 rows. Of those, 887 have non-zero `nearby_count`, the maximum is 40, and all 8 rows without coordinates remain valid. The history CSV was uploaded with the expected nine columns and 1,082 rows to `干扰历史数据/2026-08-06/干扰数据处理结果_20260806150000.csv`. - A second run `e1c81ada63234b3c85a377e1b46a1370` returned success with `status=skipped`, confirming that an existing database time plus a non-empty history file avoids duplicate work. - `MetrixApiClient` uses a proxy-free opener because the API is an intranet service and the Windows system proxy previously converted direct requests into `502` responses. The verified Metrix API remains `http://188.5.127.115:18271`; external port `9082` currently serves CapacityReport and must not be used by this script. + +## 2026-08-07: Previous-period comparison fields + +- Summary CSV, database table, and history CSV add nullable previous-period fields: `prev_interference_dbm` and `prev_nearby_count`. +- Before downloading source ZIP/XLSX files, the runner still checks the current database time. When a newer source group is being processed, it reads the currently retained database rows before replacement and maps them by CGI. The current row receives the previous row's `interference_dbm` and `nearby_count`; unmatched CGI values remain empty in CSV and `NULL` in the database. +- On the first run, when the database has no previous group, both fields remain empty/NULL. Existing-table migration adds both columns automatically. Twenty containerized tests pass, including first-run empty values, CGI matching, and nullable column migration. diff --git a/main.py b/main.py index c4c9134..722a864 100644 --- a/main.py +++ b/main.py @@ -179,10 +179,12 @@ SUMMARY_HEADER = ( "cgi", "cell_name", "interference_dbm", + "prev_interference_dbm", "longitude", "latitude", "azimuth", "nearby_count", + "prev_nearby_count", ) class ProcessingError(RuntimeError): @@ -348,7 +350,7 @@ class ApiSummaryStore: START TRANSACTION; DELETE FROM `{self.table}` WHERE metric_time = {metric_literal}; INSERT INTO `{self.table}` - (metric_time, network_type, cgi, cell_name, interference_dbm, longitude, latitude, azimuth, nearby_count) + (metric_time, network_type, cgi, cell_name, interference_dbm, prev_interference_dbm, longitude, latitude, azimuth, nearby_count, prev_nearby_count) VALUES {values}; DELETE FROM `{self.table}` WHERE metric_time <> {metric_literal}; @@ -383,7 +385,8 @@ class ApiSummaryStore: metric_literal = f"'{metric_time:%Y-%m-%d %H:%M:%S}'" sql = ( "SELECT metric_time, network_type, cgi, cell_name, interference_dbm, " - f"longitude, latitude, azimuth, nearby_count FROM `{self.table}` " + "prev_interference_dbm, longitude, latitude, azimuth, nearby_count, prev_nearby_count " + f"FROM `{self.table}` " f"WHERE metric_time = {metric_literal} ORDER BY cgi, cell_name" ) result: list[dict[str, str]] = [] @@ -415,10 +418,12 @@ class ApiSummaryStore: cgi VARCHAR(128) NOT NULL, cell_name VARCHAR(255) NOT NULL, interference_dbm DECIMAL(10,3) NOT NULL, + prev_interference_dbm DECIMAL(10,3) NULL, longitude DECIMAL(10,6) NULL, latitude DECIMAL(10,6) NULL, azimuth DECIMAL(6,2) NOT NULL DEFAULT 0, nearby_count INT NOT NULL DEFAULT 0, + prev_nearby_count INT NULL, PRIMARY KEY (metric_time, cgi) ) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4 """, @@ -433,10 +438,12 @@ class ApiSummaryStore: existing_columns = {str(item.get("name")) for item in columns if isinstance(item, dict)} migrations = { "network_type": "VARCHAR(16) NOT NULL DEFAULT ''", + "prev_interference_dbm": "DECIMAL(10,3) NULL", "longitude": "DECIMAL(10,6) NULL", "latitude": "DECIMAL(10,6) NULL", "azimuth": "DECIMAL(6,2) NOT NULL DEFAULT 0", "nearby_count": "INT NOT NULL DEFAULT 0", + "prev_nearby_count": "INT NULL", } for column, definition in migrations.items(): if column not in existing_columns: @@ -467,10 +474,12 @@ class ApiSummaryStore: sql_text_literal(row["cgi"]), sql_text_literal(row["cell_name"]), sql_decimal_literal(row["interference_dbm"], "interference_dbm"), + sql_decimal_literal(row["prev_interference_dbm"], "prev_interference_dbm", nullable=True), sql_decimal_literal(row["longitude"], "longitude", nullable=True), sql_decimal_literal(row["latitude"], "latitude", nullable=True), sql_decimal_literal(row["azimuth"], "azimuth"), str(int(row["nearby_count"])), + sql_decimal_literal(row["prev_nearby_count"], "prev_nearby_count", nullable=True), ) ) + ")" @@ -787,6 +796,8 @@ def process( print(f"missing_sources={','.join(missing)}") return None + latest_time: datetime | None = None + previous_rows: list[dict[str, str]] = [] if store is not None: latest_time = store.latest_metric_time() if latest_time is not None and latest_time >= metric_time: @@ -803,6 +814,8 @@ def process( print(f"target_time={metric_time_text}") print(f"database_metric_time={latest_time:%Y-%m-%d %H:%M:%S}") return None + if latest_time is not None: + previous_rows = store.rows_for_time(latest_time) cell_data_workbooks = source.cell_data_workbooks() cell_metadata = load_cell_metadata(cell_data_workbooks) @@ -846,6 +859,7 @@ def process( ) populate_nearby_counts(summary_rows) + populate_previous_values(summary_rows, previous_rows) summary_rows.sort(key=lambda item: (item["cgi"], item["cell_name"])) summary_name = f"interference_summary_{output_key}.csv" write_dict_csv(temp_dir / summary_name, SUMMARY_HEADER, summary_rows) @@ -941,10 +955,12 @@ def summary_record( "cgi": cgi, "cell_name": cell_name, "interference_dbm": interference, + "prev_interference_dbm": "", "longitude": longitude, "latitude": latitude, "azimuth": azimuth, "nearby_count": "0", + "prev_nearby_count": "", } @@ -989,6 +1005,18 @@ def populate_nearby_counts(rows: list[dict[str, str]]) -> None: rows[index]["nearby_count"] = str(count) +def populate_previous_values(rows: list[dict[str, str]], previous_rows: list[dict[str, str]]) -> None: + previous_by_cgi = {row["cgi"]: row for row in previous_rows} + for row in rows: + previous = previous_by_cgi.get(row["cgi"]) + if previous is None: + row["prev_interference_dbm"] = "" + row["prev_nearby_count"] = "" + continue + row["prev_interference_dbm"] = previous["interference_dbm"] + row["prev_nearby_count"] = previous["nearby_count"] + + def spatial_bucket(longitude: float, latitude: float) -> tuple[int, int, int]: longitude = math.radians(longitude) latitude = math.radians(latitude) diff --git a/tests/test_pipeline.py b/tests/test_pipeline.py index 0040eba..b0bab0c 100644 --- a/tests/test_pipeline.py +++ b/tests/test_pipeline.py @@ -49,10 +49,12 @@ class PipelineTest(unittest.TestCase): "cgi", "cell_name", "interference_dbm", + "prev_interference_dbm", "longitude", "latitude", "azimuth", "nearby_count", + "prev_nearby_count", ], ) rows_by_type = {row["cell_name"].removesuffix("-小区"): row for row in rows} @@ -65,6 +67,8 @@ class PipelineTest(unittest.TestCase): for source_type in set(main.EXPECTED_TYPES) - {"5G干扰监控", "700M干扰监控"}: self.assertEqual(rows_by_type[source_type]["cgi"], "460-00-100-1") self.assertTrue(all(row["interference_dbm"] == "-100.5" for row in rows)) + self.assertTrue(all(row["prev_interference_dbm"] == "" for row in rows)) + self.assertTrue(all(row["prev_nearby_count"] == "" for row in rows)) self.assertEqual(rows_by_type["5G干扰监控"]["longitude"], "113.123456") self.assertEqual(rows_by_type["5G干扰监控"]["latitude"], "22.654321") self.assertEqual(rows_by_type["5G干扰监控"]["azimuth"], "30") @@ -222,6 +226,22 @@ class PipelineTest(unittest.TestCase): self.assertLess(main.haversine_distance_km(113.0, 22.0, 113.005, 22.0), 1.0) self.assertGreater(main.haversine_distance_km(113.0, 22.0, 113.02, 22.0), 1.0) + def test_previous_values_are_matched_by_cgi(self) -> None: + rows = [database_row(), database_row()] + rows[0]["cgi"] = "matched" + rows[1]["cgi"] = "new" + previous = database_row() + previous["cgi"] = "matched" + previous["interference_dbm"] = "-106.25" + previous["nearby_count"] = "8" + + main.populate_previous_values(rows, [previous]) + + self.assertEqual(rows[0]["prev_interference_dbm"], "-106.25") + self.assertEqual(rows[0]["prev_nearby_count"], "8") + self.assertEqual(rows[1]["prev_interference_dbm"], "") + self.assertEqual(rows[1]["prev_nearby_count"], "") + def test_api_store_replaces_same_hour_and_deletes_other_hours(self) -> None: client = FakeApiClient("2026-07-31T10:00:00") store = main.ApiSummaryStore(client, "db_share_mysql") @@ -239,10 +259,12 @@ class PipelineTest(unittest.TestCase): self.assertTrue(any("ADD COLUMN `latitude`" in query for query in queries)) self.assertTrue(any("ADD COLUMN `azimuth`" in query for query in queries)) self.assertTrue(any("ADD COLUMN `nearby_count`" in query for query in queries)) + self.assertTrue(any("ADD COLUMN `prev_interference_dbm`" in query for query in queries)) + self.assertTrue(any("ADD COLUMN `prev_nearby_count`" in query for query in queries)) script = next(payload for endpoint, payload in client.posts if endpoint.endswith("/run-script")) self.assertTrue(script["single_session"]) self.assertIn("START TRANSACTION", script["content"]) - self.assertIn("NULL, NULL, 0, 0", script["content"]) + self.assertIn("NULL, NULL, NULL, 0, 0, NULL", script["content"]) self.assertIn("CONVERT(0x", script["content"]) self.assertNotIn("source_type", script["content"]) self.assertNotIn("source_path", script["content"]) @@ -447,10 +469,12 @@ def database_row() -> dict[str, str]: "cgi": "460-00-200-1", "cell_name": "测试小区", "interference_dbm": "-100.5", + "prev_interference_dbm": "", "longitude": "", "latitude": "", "azimuth": "0", "nearby_count": "0", + "prev_nearby_count": "", }