fix: 按最大干扰值去重CGI
This commit is contained in:
@@ -34,6 +34,8 @@ All selected workbook rows must belong to the selected natural 15-minute group.
|
||||
|
||||
`nearby_count` is the number of other high-interference cells from the same selected time group within 1 km, across all network types. The cell itself is excluded, the 1 km boundary is included, and rows without coordinates use `0`. Candidate cells are found through a 1 km spatial grid index and confirmed with exact Haversine distance.
|
||||
|
||||
When multiple retained source rows have the same CGI, the summary keeps the row with the numerically largest interference value. Equal values keep the first row encountered. Deduplication happens before nearby-cell counting and database insertion; converted source CSVs remain complete.
|
||||
|
||||
`prev_interference_dbm` and `prev_nearby_count` come from the previously retained database time, matched by CGI before the current batch replaces old rows. They represent the same cell's previous-period interference value and nearby high-interference count. On the first run, or when a CGI did not exist in the previous period, both fields are empty in CSV/history output and `NULL` in MySQL.
|
||||
|
||||
The script never modifies or deletes source storage files. Before downloading source ZIP files or CellData, it compares the normalized target time with the database. A target already present, or older than the database, is not processed again.
|
||||
|
||||
@@ -108,3 +108,8 @@
|
||||
- Twenty-one containerized tests and offline-package execution pass. The deployed package SHA-256 is `11B0010E8FDFEC32058BC8B282E5A5608FC92C48572611A3F9F4632962C18D98`.
|
||||
- The current `2026-08-07 14:00:00` database result was exported directly through the history-only path and replaced in Storage as a verified GBK CSV: 1,082 rows, 150,251 bytes, no UTF-8 BOM, and the expected eleven columns. The temporary UTF-8 backup was deleted after validation.
|
||||
- The newer `2026-08-07 15:00:00` source group currently contains duplicate CGI rows, for example `460-00-122737-22`. Its database transaction rolled back on the existing `(metric_time, cgi)` primary key, so the retained `14:00` database and history result were not replaced. Duplicate-row selection requires a separate business rule.
|
||||
|
||||
## 2026-08-07: Duplicate CGI selection
|
||||
|
||||
- Retained high-interference rows are deduplicated by CGI before nearby counting and database insertion. When duplicates exist, the row with the numerically largest `interference_dbm` is kept; equal values keep the first encountered row. Converted source CSVs remain complete.
|
||||
- The deduplication count is recorded as `duplicate_cgi_rows` in `manifest.json` and stdout. This prevents the existing `(metric_time, cgi)` database primary key from failing and prevents duplicate source rows from inflating `nearby_count`.
|
||||
|
||||
@@ -863,6 +863,7 @@ def process(
|
||||
}
|
||||
)
|
||||
|
||||
duplicate_cgi_rows = deduplicate_by_cgi(summary_rows)
|
||||
populate_nearby_counts(summary_rows)
|
||||
populate_previous_values(summary_rows, previous_rows)
|
||||
summary_rows.sort(key=lambda item: (item["cgi"], item["cell_name"]))
|
||||
@@ -884,6 +885,7 @@ def process(
|
||||
"cell_data_file_count": len(cell_data_workbooks),
|
||||
"summary_rows": len(summary_rows),
|
||||
"threshold_filtered_rows": threshold_filtered_rows,
|
||||
"duplicate_cgi_rows": duplicate_cgi_rows,
|
||||
"coordinate_matched_rows": matched_coordinates,
|
||||
"coordinate_unmatched_rows": len(summary_rows) - matched_coordinates,
|
||||
"files": manifest_files,
|
||||
@@ -907,6 +909,7 @@ def process(
|
||||
print(f"cell_data_files={len(cell_data_workbooks)}")
|
||||
print(f"summary_rows={len(summary_rows)}")
|
||||
print(f"threshold_filtered_rows={threshold_filtered_rows}")
|
||||
print(f"duplicate_cgi_rows={duplicate_cgi_rows}")
|
||||
print(f"coordinate_matched_rows={matched_coordinates}")
|
||||
print(f"coordinate_unmatched_rows={len(summary_rows) - matched_coordinates}")
|
||||
if database_result["enabled"]:
|
||||
@@ -1013,6 +1016,21 @@ def populate_nearby_counts(rows: list[dict[str, str]]) -> None:
|
||||
rows[index]["nearby_count"] = str(count)
|
||||
|
||||
|
||||
def deduplicate_by_cgi(rows: list[dict[str, str]]) -> int:
|
||||
selected: dict[str, dict[str, str]] = {}
|
||||
removed = 0
|
||||
for row in rows:
|
||||
previous = selected.get(row["cgi"])
|
||||
if previous is None:
|
||||
selected[row["cgi"]] = row
|
||||
continue
|
||||
if Decimal(row["interference_dbm"]) > Decimal(previous["interference_dbm"]):
|
||||
selected[row["cgi"]] = row
|
||||
removed += 1
|
||||
rows[:] = selected.values()
|
||||
return removed
|
||||
|
||||
|
||||
def populate_previous_values(rows: list[dict[str, str]], previous_rows: list[dict[str, str]]) -> None:
|
||||
previous_by_cgi = {row["cgi"]: row for row in previous_rows}
|
||||
for row in rows:
|
||||
|
||||
+36
-23
@@ -43,7 +43,7 @@ class PipelineTest(unittest.TestCase):
|
||||
self.assertTrue(summary_bytes.startswith(b"\xef\xbb\xbf"))
|
||||
with summary_path.open(encoding="utf-8-sig", newline="") as file:
|
||||
rows = list(csv.DictReader(file))
|
||||
self.assertEqual(len(rows), 7)
|
||||
self.assertEqual(len(rows), 2)
|
||||
self.assertEqual(
|
||||
list(rows[0]),
|
||||
[
|
||||
@@ -60,34 +60,29 @@ class PipelineTest(unittest.TestCase):
|
||||
"prev_nearby_count",
|
||||
],
|
||||
)
|
||||
rows_by_type = {row["cell_name"].removesuffix("-小区"): row for row in rows}
|
||||
self.assertEqual(rows_by_type["5G干扰监控"]["network_type"], "2.6G")
|
||||
self.assertEqual(rows_by_type["700M干扰监控"]["network_type"], "700M")
|
||||
self.assertEqual(rows_by_type["SDR_FDD干扰监控"]["network_type"], "4G")
|
||||
rows_by_cgi = {row["cgi"]: row for row in rows}
|
||||
self.assertEqual(rows_by_cgi["460-00-200-1"]["network_type"], "2.6G")
|
||||
self.assertTrue(all(row["metric_time"] == "2026-07-31 10:00:00" for row in rows))
|
||||
self.assertEqual(rows_by_type["5G干扰监控"]["cgi"], "460-00-200-1")
|
||||
self.assertEqual(rows_by_type["700M干扰监控"]["cgi"], "460-00-200-1")
|
||||
for source_type in set(main.EXPECTED_TYPES) - {"5G干扰监控", "700M干扰监控"}:
|
||||
self.assertEqual(rows_by_type[source_type]["cgi"], "460-00-100-1")
|
||||
self.assertEqual(rows_by_cgi["460-00-200-1"]["cell_name"], "5G干扰监控-小区")
|
||||
self.assertTrue(all(row["interference_dbm"] == "-100.5" for row in rows))
|
||||
self.assertTrue(all(row["prev_interference_dbm"] == "" for row in rows))
|
||||
self.assertTrue(all(row["prev_nearby_count"] == "" for row in rows))
|
||||
self.assertEqual(rows_by_type["5G干扰监控"]["longitude"], "113.123456")
|
||||
self.assertEqual(rows_by_type["5G干扰监控"]["latitude"], "22.654321")
|
||||
self.assertEqual(rows_by_type["5G干扰监控"]["azimuth"], "30")
|
||||
self.assertEqual(rows_by_type["5G干扰监控"]["nearby_count"], "1")
|
||||
self.assertEqual(rows_by_type["700M干扰监控"]["nearby_count"], "1")
|
||||
self.assertTrue(all(not row["longitude"] and not row["latitude"] for row in rows if row["cgi"] == "460-00-100-1"))
|
||||
self.assertTrue(all(row["azimuth"] == "0" for row in rows if row["cgi"] == "460-00-100-1"))
|
||||
self.assertTrue(all(row["nearby_count"] == "0" for row in rows if row["cgi"] == "460-00-100-1"))
|
||||
self.assertEqual(rows_by_cgi["460-00-200-1"]["longitude"], "113.123456")
|
||||
self.assertEqual(rows_by_cgi["460-00-200-1"]["latitude"], "22.654321")
|
||||
self.assertEqual(rows_by_cgi["460-00-200-1"]["azimuth"], "30")
|
||||
self.assertEqual(rows_by_cgi["460-00-200-1"]["nearby_count"], "0")
|
||||
self.assertEqual(rows_by_cgi["460-00-100-1"]["nearby_count"], "0")
|
||||
self.assertFalse(rows_by_cgi["460-00-100-1"]["longitude"])
|
||||
self.assertEqual(rows_by_cgi["460-00-100-1"]["azimuth"], "0")
|
||||
|
||||
manifest = json.loads((result / "manifest.json").read_text(encoding="utf-8"))
|
||||
self.assertEqual(manifest["source_file_count"], 7)
|
||||
self.assertEqual(manifest["cell_data_file_count"], 5)
|
||||
self.assertEqual(manifest["summary_rows"], 7)
|
||||
self.assertEqual(manifest["summary_rows"], 2)
|
||||
self.assertEqual(manifest["threshold_filtered_rows"], 0)
|
||||
self.assertEqual(manifest["coordinate_matched_rows"], 2)
|
||||
self.assertEqual(manifest["coordinate_unmatched_rows"], 5)
|
||||
self.assertEqual(manifest["duplicate_cgi_rows"], 5)
|
||||
self.assertEqual(manifest["coordinate_matched_rows"], 1)
|
||||
self.assertEqual(manifest["coordinate_unmatched_rows"], 1)
|
||||
self.assertEqual(manifest["metric_time"], "2026-07-31 10:00:00")
|
||||
self.assertNotIn("hour_start", manifest)
|
||||
self.assertNotIn("hour_end", manifest)
|
||||
@@ -95,11 +90,11 @@ class PipelineTest(unittest.TestCase):
|
||||
self.assertTrue(all("source_type" not in item and "source_path" not in item for item in manifest["files"]))
|
||||
self.assertIn("2026073110091100", [item["source_window"] for item in manifest["files"]])
|
||||
self.assertEqual(manifest["database"]["metric_time"], "2026-07-31 10:00:00")
|
||||
self.assertEqual(len(store.rows), 7)
|
||||
self.assertEqual(len(store.rows), 2)
|
||||
self.assertEqual(history.path, "/history/2026-07-31/干扰数据处理结果_20260731100000.csv")
|
||||
archived_rows = list(csv.DictReader(io.StringIO(history.payload.decode("gbk"))))
|
||||
self.assertEqual(list(archived_rows[0]), list(main.SUMMARY_HEADER))
|
||||
self.assertEqual(len(archived_rows), 7)
|
||||
self.assertEqual(len(archived_rows), 2)
|
||||
|
||||
def test_interference_thresholds_remove_only_lower_values(self) -> None:
|
||||
for network_type, threshold in (("2.6G", "-107"), ("700M", "-110"), ("4G", "-110")):
|
||||
@@ -120,8 +115,9 @@ class PipelineTest(unittest.TestCase):
|
||||
result = main.process(main.LocalSource(root), Path(temp) / "output", 3)
|
||||
manifest = json.loads((result / "manifest.json").read_text(encoding="utf-8"))
|
||||
|
||||
self.assertEqual(manifest["summary_rows"], 6)
|
||||
self.assertEqual(manifest["summary_rows"], 2)
|
||||
self.assertEqual(manifest["threshold_filtered_rows"], 1)
|
||||
self.assertEqual(manifest["duplicate_cgi_rows"], 4)
|
||||
|
||||
def test_latest_incomplete_group_waits_without_falling_back(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as temp:
|
||||
@@ -245,6 +241,23 @@ class PipelineTest(unittest.TestCase):
|
||||
self.assertEqual(rows[1]["prev_interference_dbm"], "")
|
||||
self.assertEqual(rows[1]["prev_nearby_count"], "")
|
||||
|
||||
def test_duplicate_cgi_keeps_maximum_interference(self) -> None:
|
||||
rows = [database_row(), database_row(), database_row()]
|
||||
rows[0]["cgi"] = "duplicate"
|
||||
rows[0]["interference_dbm"] = "-108.5"
|
||||
rows[1]["cgi"] = "duplicate"
|
||||
rows[1]["interference_dbm"] = "-101.25"
|
||||
rows[2]["cgi"] = "unique"
|
||||
|
||||
removed = main.deduplicate_by_cgi(rows)
|
||||
|
||||
self.assertEqual(removed, 1)
|
||||
self.assertEqual(len(rows), 2)
|
||||
self.assertEqual({row["cgi"]: row["interference_dbm"] for row in rows}, {
|
||||
"duplicate": "-101.25",
|
||||
"unique": "-100.5",
|
||||
})
|
||||
|
||||
def test_gbk_history_csv_normalizes_non_breaking_spaces(self) -> None:
|
||||
payload = main.dict_csv_bytes(
|
||||
("cell_name",),
|
||||
|
||||
Reference in New Issue
Block a user