fix: 兼容GBK不间断空格
This commit is contained in:
@@ -48,7 +48,7 @@ After the database succeeds, the same eleven result columns are uploaded as GBK
|
|||||||
|
|
||||||
The filename time comes from normalized `metric_time`, never the server clock. If the database already contains the target time but its history CSV is missing or empty, the script exports that time from the database and repairs the history file without reprocessing source data.
|
The filename time comes from normalized `metric_time`, never the server clock. If the database already contains the target time but its history CSV is missing or empty, the script exports that time from the database and repairs the history file without reprocessing source data.
|
||||||
|
|
||||||
All generated CSV files use GBK without a UTF-8 BOM, including the seven converted source files, the merged summary, and the uploaded history file. `manifest.json` remains UTF-8 JSON.
|
All generated CSV files use GBK without a UTF-8 BOM, including the seven converted source files, the merged summary, and the uploaded history file. Source non-breaking spaces (`U+00A0`) are normalized to ordinary spaces because GBK cannot encode them; other unsupported characters still fail the run instead of being silently replaced. `manifest.json` remains UTF-8 JSON.
|
||||||
|
|
||||||
CGI is generated with fixed rules:
|
CGI is generated with fixed rules:
|
||||||
|
|
||||||
|
|||||||
@@ -104,3 +104,4 @@
|
|||||||
|
|
||||||
- All generated CSV files now use GBK without a UTF-8 BOM: the seven full source conversions, the merged summary, repaired history exports, and uploaded history files. `manifest.json`, API JSON, and database character sets remain unchanged.
|
- All generated CSV files now use GBK without a UTF-8 BOM: the seven full source conversions, the merged summary, repaired history exports, and uploaded history files. `manifest.json`, API JSON, and database character sets remain unchanged.
|
||||||
- The encoding is centralized in `CSV_ENCODING`; tests verify GBK Chinese bytes, absence of the UTF-8 BOM, summary parsing, and history parsing.
|
- The encoding is centralized in `CSV_ENCODING`; tests verify GBK Chinese bytes, absence of the UTF-8 BOM, summary parsing, and history parsing.
|
||||||
|
- Production data contains non-breaking spaces (`U+00A0`), which Python's GBK codec cannot encode. CSV-only normalization converts them to ordinary spaces while keeping strict encoding for every other unsupported character, so unexpected data still fails visibly instead of losing text silently.
|
||||||
|
|||||||
+1
-1
@@ -8,7 +8,7 @@
|
|||||||
- Only `Sheet0` is converted and merged. The `指标(计数器)` sheet is metadata and is not included in the summary.
|
- Only `Sheet0` is converted and merged. The `指标(计数器)` sheet is metadata and is not included in the summary.
|
||||||
- NR CGI is derived as `{gNBplmn}-{gNBId}-{cellId}`.
|
- NR CGI is derived as `{gNBplmn}-{gNBId}-{cellId}`.
|
||||||
- 4G CGI is derived as `460-00-{eNodeBID}-{小区ID}`, using the equivalent node and cell column names in each source schema.
|
- 4G CGI is derived as `460-00-{eNodeBID}-{小区ID}`, using the equivalent node and cell column names in each source schema.
|
||||||
- All output CSV files use GBK without a UTF-8 BOM. `manifest.json` remains UTF-8 JSON.
|
- All output CSV files use GBK without a UTF-8 BOM. Non-breaking spaces are normalized to ordinary spaces; other unsupported characters remain encoding errors. `manifest.json` remains UTF-8 JSON.
|
||||||
- Source files are read-only. The script writes only below its configured output directory.
|
- Source files are read-only. The script writes only below its configured output directory.
|
||||||
- MySQL is the scheduled-run output. The table uses one `metric_time DATETIME` column containing the source KPI start time.
|
- MySQL is the scheduled-run output. The table uses one `metric_time DATETIME` column containing the source KPI start time.
|
||||||
- A successful run transactionally refreshes the selected hour and deletes rows for all other hours. A selected hour older than the newest database hour is rejected to prevent data rollback.
|
- A successful run transactionally refreshes the selected hour and deletes rows for all other hours. A selected hour older than the newest database hour is rejected to prevent data rollback.
|
||||||
|
|||||||
@@ -1064,6 +1064,10 @@ def normalize_cell(value: object) -> str:
|
|||||||
return str(value).strip()
|
return str(value).strip()
|
||||||
|
|
||||||
|
|
||||||
|
def normalize_csv_cell(value: object) -> str:
|
||||||
|
return normalize_cell(value).replace("\u00a0", " ")
|
||||||
|
|
||||||
|
|
||||||
def sql_text_literal(value: str) -> str:
|
def sql_text_literal(value: str) -> str:
|
||||||
encoded = value.encode("utf-8").hex()
|
encoded = value.encode("utf-8").hex()
|
||||||
return "''" if not encoded else f"CONVERT(0x{encoded} USING utf8mb4)"
|
return "''" if not encoded else f"CONVERT(0x{encoded} USING utf8mb4)"
|
||||||
@@ -1087,7 +1091,7 @@ def write_csv(path: Path, header: tuple[str, ...], rows: list[tuple[object, ...]
|
|||||||
writer = csv.writer(file)
|
writer = csv.writer(file)
|
||||||
writer.writerow(header)
|
writer.writerow(header)
|
||||||
for row in rows:
|
for row in rows:
|
||||||
writer.writerow(normalize_cell(value) for value in row)
|
writer.writerow(normalize_csv_cell(value) for value in row)
|
||||||
|
|
||||||
|
|
||||||
def write_dict_csv(path: Path, header: tuple[str, ...], rows: list[dict[str, str]]) -> None:
|
def write_dict_csv(path: Path, header: tuple[str, ...], rows: list[dict[str, str]]) -> None:
|
||||||
@@ -1098,7 +1102,8 @@ def dict_csv_bytes(header: tuple[str, ...], rows: list[dict[str, str]]) -> bytes
|
|||||||
output = io.StringIO(newline="")
|
output = io.StringIO(newline="")
|
||||||
writer = csv.DictWriter(output, fieldnames=header, extrasaction="raise")
|
writer = csv.DictWriter(output, fieldnames=header, extrasaction="raise")
|
||||||
writer.writeheader()
|
writer.writeheader()
|
||||||
writer.writerows(rows)
|
for row in rows:
|
||||||
|
writer.writerow({column: normalize_csv_cell(row[column]) for column in header})
|
||||||
return output.getvalue().encode(CSV_ENCODING)
|
return output.getvalue().encode(CSV_ENCODING)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -246,6 +246,11 @@ class PipelineTest(unittest.TestCase):
|
|||||||
self.assertEqual(rows[1]["prev_interference_dbm"], "")
|
self.assertEqual(rows[1]["prev_interference_dbm"], "")
|
||||||
self.assertEqual(rows[1]["prev_nearby_count"], "")
|
self.assertEqual(rows[1]["prev_nearby_count"], "")
|
||||||
|
|
||||||
|
def test_gbk_csv_normalizes_non_breaking_spaces(self) -> None:
|
||||||
|
payload = main.dict_csv_bytes(("cell_name",), [{"cell_name": "测试\u00a0小区"}])
|
||||||
|
|
||||||
|
self.assertEqual(payload.decode("gbk"), "cell_name\r\n测试 小区\r\n")
|
||||||
|
|
||||||
def test_api_store_replaces_same_hour_and_deletes_other_hours(self) -> None:
|
def test_api_store_replaces_same_hour_and_deletes_other_hours(self) -> None:
|
||||||
client = FakeApiClient("2026-07-31T10:00:00")
|
client = FakeApiClient("2026-07-31T10:00:00")
|
||||||
store = main.ApiSummaryStore(client, "db_share_mysql")
|
store = main.ApiSummaryStore(client, "db_share_mysql")
|
||||||
|
|||||||
Reference in New Issue
Block a user