feat: 增加上一周期干扰对比字段

This commit is contained in:
2026-08-07 09:22:26 +08:00
parent 1c752e0f75
commit 5e4b58255f
4 changed files with 65 additions and 5 deletions
+4 -2
View File
@@ -25,7 +25,7 @@ For every processed group, the script also reads the latest filename-dated XLSX
The summary columns are:
```text
metric_time,network_type,cgi,cell_name,interference_dbm,longitude,latitude,azimuth,nearby_count
metric_time,network_type,cgi,cell_name,interference_dbm,prev_interference_dbm,longitude,latitude,azimuth,nearby_count,prev_nearby_count
```
All selected workbook rows must belong to the selected natural 15-minute group. Every summary row receives the same normalized `metric_time`; a row outside the group fails the run before database or history output.
@@ -34,9 +34,11 @@ All selected workbook rows must belong to the selected natural 15-minute group.
`nearby_count` is the number of other high-interference cells from the same selected time group within 1 km, across all network types. The cell itself is excluded, the 1 km boundary is included, and rows without coordinates use `0`. Candidate cells are found through a 1 km spatial grid index and confirmed with exact Haversine distance.
`prev_interference_dbm` and `prev_nearby_count` come from the previously retained database time, matched by CGI before the current batch replaces old rows. They represent the same cell's previous-period interference value and nearby high-interference count. On the first run, or when a CGI did not exist in the previous period, both fields are empty in CSV/history output and `NULL` in MySQL.
The script never modifies or deletes source storage files. Before downloading source ZIP files or CellData, it compares the normalized target time with the database. A target already present, or older than the database, is not processed again.
The database table contains normalized `metric_time DATETIME`, `network_type VARCHAR(16)`, nullable `longitude` and `latitude`, `azimuth DECIMAL(6,2) NOT NULL DEFAULT 0`, and `nearby_count INT NOT NULL DEFAULT 0`. A successful transaction replaces the target batch and deletes every other database time, so the table retains only the latest processed group.
The database table contains normalized `metric_time DATETIME`, `network_type VARCHAR(16)`, `interference_dbm DECIMAL(10,3)`, nullable `prev_interference_dbm DECIMAL(10,3)`, nullable `longitude` and `latitude`, `azimuth DECIMAL(6,2) NOT NULL DEFAULT 0`, `nearby_count INT NOT NULL DEFAULT 0`, and nullable `prev_nearby_count INT`. A successful transaction replaces the target batch and deletes every other database time, so the table retains only the latest processed group. The previous-period values are read before this replacement transaction.
After the database succeeds, the same nine result columns are uploaded as UTF-8-BOM CSV to:
+6
View File
@@ -91,3 +91,9 @@
- Database verification found only `2026-08-06 15:00:00` and 1,082 rows. Of those, 887 have non-zero `nearby_count`, the maximum is 40, and all 8 rows without coordinates remain valid. The history CSV was uploaded with the expected nine columns and 1,082 rows to `干扰历史数据/2026-08-06/干扰数据处理结果_20260806150000.csv`.
- A second run `e1c81ada63234b3c85a377e1b46a1370` returned success with `status=skipped`, confirming that an existing database time plus a non-empty history file avoids duplicate work.
- `MetrixApiClient` uses a proxy-free opener because the API is an intranet service and the Windows system proxy previously converted direct requests into `502` responses. The verified Metrix API remains `http://188.5.127.115:18271`; external port `9082` currently serves CapacityReport and must not be used by this script.
## 2026-08-07: Previous-period comparison fields
- Summary CSV, database table, and history CSV add nullable previous-period fields: `prev_interference_dbm` and `prev_nearby_count`.
- Before downloading source ZIP/XLSX files, the runner still checks the current database time. When a newer source group is being processed, it reads the currently retained database rows before replacement and maps them by CGI. The current row receives the previous row's `interference_dbm` and `nearby_count`; unmatched CGI values remain empty in CSV and `NULL` in the database.
- On the first run, when the database has no previous group, both fields remain empty/NULL. Existing-table migration adds both columns automatically. Twenty containerized tests pass, including first-run empty values, CGI matching, and nullable column migration.
+30 -2
View File
@@ -179,10 +179,12 @@ SUMMARY_HEADER = (
"cgi",
"cell_name",
"interference_dbm",
"prev_interference_dbm",
"longitude",
"latitude",
"azimuth",
"nearby_count",
"prev_nearby_count",
)
class ProcessingError(RuntimeError):
@@ -348,7 +350,7 @@ class ApiSummaryStore:
START TRANSACTION;
DELETE FROM `{self.table}` WHERE metric_time = {metric_literal};
INSERT INTO `{self.table}`
(metric_time, network_type, cgi, cell_name, interference_dbm, longitude, latitude, azimuth, nearby_count)
(metric_time, network_type, cgi, cell_name, interference_dbm, prev_interference_dbm, longitude, latitude, azimuth, nearby_count, prev_nearby_count)
VALUES
{values};
DELETE FROM `{self.table}` WHERE metric_time <> {metric_literal};
@@ -383,7 +385,8 @@ class ApiSummaryStore:
metric_literal = f"'{metric_time:%Y-%m-%d %H:%M:%S}'"
sql = (
"SELECT metric_time, network_type, cgi, cell_name, interference_dbm, "
f"longitude, latitude, azimuth, nearby_count FROM `{self.table}` "
"prev_interference_dbm, longitude, latitude, azimuth, nearby_count, prev_nearby_count "
f"FROM `{self.table}` "
f"WHERE metric_time = {metric_literal} ORDER BY cgi, cell_name"
)
result: list[dict[str, str]] = []
@@ -415,10 +418,12 @@ class ApiSummaryStore:
cgi VARCHAR(128) NOT NULL,
cell_name VARCHAR(255) NOT NULL,
interference_dbm DECIMAL(10,3) NOT NULL,
prev_interference_dbm DECIMAL(10,3) NULL,
longitude DECIMAL(10,6) NULL,
latitude DECIMAL(10,6) NULL,
azimuth DECIMAL(6,2) NOT NULL DEFAULT 0,
nearby_count INT NOT NULL DEFAULT 0,
prev_nearby_count INT NULL,
PRIMARY KEY (metric_time, cgi)
) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4
""",
@@ -433,10 +438,12 @@ class ApiSummaryStore:
existing_columns = {str(item.get("name")) for item in columns if isinstance(item, dict)}
migrations = {
"network_type": "VARCHAR(16) NOT NULL DEFAULT ''",
"prev_interference_dbm": "DECIMAL(10,3) NULL",
"longitude": "DECIMAL(10,6) NULL",
"latitude": "DECIMAL(10,6) NULL",
"azimuth": "DECIMAL(6,2) NOT NULL DEFAULT 0",
"nearby_count": "INT NOT NULL DEFAULT 0",
"prev_nearby_count": "INT NULL",
}
for column, definition in migrations.items():
if column not in existing_columns:
@@ -467,10 +474,12 @@ class ApiSummaryStore:
sql_text_literal(row["cgi"]),
sql_text_literal(row["cell_name"]),
sql_decimal_literal(row["interference_dbm"], "interference_dbm"),
sql_decimal_literal(row["prev_interference_dbm"], "prev_interference_dbm", nullable=True),
sql_decimal_literal(row["longitude"], "longitude", nullable=True),
sql_decimal_literal(row["latitude"], "latitude", nullable=True),
sql_decimal_literal(row["azimuth"], "azimuth"),
str(int(row["nearby_count"])),
sql_decimal_literal(row["prev_nearby_count"], "prev_nearby_count", nullable=True),
)
) + ")"
@@ -787,6 +796,8 @@ def process(
print(f"missing_sources={','.join(missing)}")
return None
latest_time: datetime | None = None
previous_rows: list[dict[str, str]] = []
if store is not None:
latest_time = store.latest_metric_time()
if latest_time is not None and latest_time >= metric_time:
@@ -803,6 +814,8 @@ def process(
print(f"target_time={metric_time_text}")
print(f"database_metric_time={latest_time:%Y-%m-%d %H:%M:%S}")
return None
if latest_time is not None:
previous_rows = store.rows_for_time(latest_time)
cell_data_workbooks = source.cell_data_workbooks()
cell_metadata = load_cell_metadata(cell_data_workbooks)
@@ -846,6 +859,7 @@ def process(
)
populate_nearby_counts(summary_rows)
populate_previous_values(summary_rows, previous_rows)
summary_rows.sort(key=lambda item: (item["cgi"], item["cell_name"]))
summary_name = f"interference_summary_{output_key}.csv"
write_dict_csv(temp_dir / summary_name, SUMMARY_HEADER, summary_rows)
@@ -941,10 +955,12 @@ def summary_record(
"cgi": cgi,
"cell_name": cell_name,
"interference_dbm": interference,
"prev_interference_dbm": "",
"longitude": longitude,
"latitude": latitude,
"azimuth": azimuth,
"nearby_count": "0",
"prev_nearby_count": "",
}
@@ -989,6 +1005,18 @@ def populate_nearby_counts(rows: list[dict[str, str]]) -> None:
rows[index]["nearby_count"] = str(count)
def populate_previous_values(rows: list[dict[str, str]], previous_rows: list[dict[str, str]]) -> None:
previous_by_cgi = {row["cgi"]: row for row in previous_rows}
for row in rows:
previous = previous_by_cgi.get(row["cgi"])
if previous is None:
row["prev_interference_dbm"] = ""
row["prev_nearby_count"] = ""
continue
row["prev_interference_dbm"] = previous["interference_dbm"]
row["prev_nearby_count"] = previous["nearby_count"]
def spatial_bucket(longitude: float, latitude: float) -> tuple[int, int, int]:
longitude = math.radians(longitude)
latitude = math.radians(latitude)
+25 -1
View File
@@ -49,10 +49,12 @@ class PipelineTest(unittest.TestCase):
"cgi",
"cell_name",
"interference_dbm",
"prev_interference_dbm",
"longitude",
"latitude",
"azimuth",
"nearby_count",
"prev_nearby_count",
],
)
rows_by_type = {row["cell_name"].removesuffix("-小区"): row for row in rows}
@@ -65,6 +67,8 @@ class PipelineTest(unittest.TestCase):
for source_type in set(main.EXPECTED_TYPES) - {"5G干扰监控", "700M干扰监控"}:
self.assertEqual(rows_by_type[source_type]["cgi"], "460-00-100-1")
self.assertTrue(all(row["interference_dbm"] == "-100.5" for row in rows))
self.assertTrue(all(row["prev_interference_dbm"] == "" for row in rows))
self.assertTrue(all(row["prev_nearby_count"] == "" for row in rows))
self.assertEqual(rows_by_type["5G干扰监控"]["longitude"], "113.123456")
self.assertEqual(rows_by_type["5G干扰监控"]["latitude"], "22.654321")
self.assertEqual(rows_by_type["5G干扰监控"]["azimuth"], "30")
@@ -222,6 +226,22 @@ class PipelineTest(unittest.TestCase):
self.assertLess(main.haversine_distance_km(113.0, 22.0, 113.005, 22.0), 1.0)
self.assertGreater(main.haversine_distance_km(113.0, 22.0, 113.02, 22.0), 1.0)
def test_previous_values_are_matched_by_cgi(self) -> None:
rows = [database_row(), database_row()]
rows[0]["cgi"] = "matched"
rows[1]["cgi"] = "new"
previous = database_row()
previous["cgi"] = "matched"
previous["interference_dbm"] = "-106.25"
previous["nearby_count"] = "8"
main.populate_previous_values(rows, [previous])
self.assertEqual(rows[0]["prev_interference_dbm"], "-106.25")
self.assertEqual(rows[0]["prev_nearby_count"], "8")
self.assertEqual(rows[1]["prev_interference_dbm"], "")
self.assertEqual(rows[1]["prev_nearby_count"], "")
def test_api_store_replaces_same_hour_and_deletes_other_hours(self) -> None:
client = FakeApiClient("2026-07-31T10:00:00")
store = main.ApiSummaryStore(client, "db_share_mysql")
@@ -239,10 +259,12 @@ class PipelineTest(unittest.TestCase):
self.assertTrue(any("ADD COLUMN `latitude`" in query for query in queries))
self.assertTrue(any("ADD COLUMN `azimuth`" in query for query in queries))
self.assertTrue(any("ADD COLUMN `nearby_count`" in query for query in queries))
self.assertTrue(any("ADD COLUMN `prev_interference_dbm`" in query for query in queries))
self.assertTrue(any("ADD COLUMN `prev_nearby_count`" in query for query in queries))
script = next(payload for endpoint, payload in client.posts if endpoint.endswith("/run-script"))
self.assertTrue(script["single_session"])
self.assertIn("START TRANSACTION", script["content"])
self.assertIn("NULL, NULL, 0, 0", script["content"])
self.assertIn("NULL, NULL, NULL, 0, 0, NULL", script["content"])
self.assertIn("CONVERT(0x", script["content"])
self.assertNotIn("source_type", script["content"])
self.assertNotIn("source_path", script["content"])
@@ -447,10 +469,12 @@ def database_row() -> dict[str, str]:
"cgi": "460-00-200-1",
"cell_name": "测试小区",
"interference_dbm": "-100.5",
"prev_interference_dbm": "",
"longitude": "",
"latitude": "",
"azimuth": "0",
"nearby_count": "0",
"prev_nearby_count": "",
}