From 8d681118ea443aa2ecf94fedfeabb88517e76c9c Mon Sep 17 00:00:00 2001 From: Nixevol Date: Fri, 7 Aug 2026 16:00:30 +0800 Subject: [PATCH] =?UTF-8?q?feat:=20=E5=B0=86=E5=B9=B2=E6=89=B0CSV=E8=BE=93?= =?UTF-8?q?=E5=87=BA=E6=94=B9=E4=B8=BAGBK?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- README.md | 4 +++- aidocs/project_context.md | 5 +++++ docs/assumptions.md | 2 +- main.py | 5 +++-- tests/test_pipeline.py | 10 +++++++--- 5 files changed, 19 insertions(+), 7 deletions(-) diff --git a/README.md b/README.md index ac37d20..efbc154 100644 --- a/README.md +++ b/README.md @@ -40,7 +40,7 @@ The script never modifies or deletes source storage files. Before downloading so The database table contains normalized `metric_time DATETIME`, `network_type VARCHAR(16)`, `interference_dbm DECIMAL(10,3)`, nullable `prev_interference_dbm DECIMAL(10,3)`, nullable `longitude` and `latitude`, `azimuth DECIMAL(6,2) NOT NULL DEFAULT 0`, `nearby_count INT NOT NULL DEFAULT 0`, and nullable `prev_nearby_count INT`. A successful transaction replaces the target batch and deletes every other database time, so the table retains only the latest processed group. The previous-period values are read before this replacement transaction. -After the database succeeds, the same nine result columns are uploaded as UTF-8-BOM CSV to: +After the database succeeds, the same eleven result columns are uploaded as GBK CSV to: ```text /网优日常优化数据文档/(勿删)干扰定时小时指标/干扰历史数据/YYYY-MM-DD/干扰数据处理结果_YYYYMMDDHHMMSS.csv @@ -48,6 +48,8 @@ After the database succeeds, the same nine result columns are uploaded as UTF-8- The filename time comes from normalized `metric_time`, never the server clock. If the database already contains the target time but its history CSV is missing or empty, the script exports that time from the database and repairs the history file without reprocessing source data. +All generated CSV files use GBK without a UTF-8 BOM, including the seven converted source files, the merged summary, and the uploaded history file. `manifest.json` remains UTF-8 JSON. + CGI is generated with fixed rules: - NR (`5G干扰监控`, `700M干扰监控`): `{gNBplmn}-{gNBId}-{cellId}`. diff --git a/aidocs/project_context.md b/aidocs/project_context.md index 668eb33..a343823 100644 --- a/aidocs/project_context.md +++ b/aidocs/project_context.md @@ -99,3 +99,8 @@ - On the first run, when the database has no previous group, both fields remain empty/NULL. Existing-table migration adds both columns automatically. Twenty containerized tests pass, including first-run empty values, CGI matching, and nullable column migration. - Offline package SHA-256 is `74EEA84058C5F14E7108B840561C070DC47752D67FEC68262F0CC6C78EE8EC0D`. SSH deployment run `984ffc6f98024a2d81b63d4359aabc71` processed `2026-08-07 07:00:00` with 820 rows and deleted 1,137 rows from the prior database time. - Production verification matched 616 current CGI values to the previous period and left both previous fields NULL for the remaining 204 CGI values. `prev_nearby_count` matched exactly; `prev_interference_dbm` uses the table's existing `DECIMAL(10,3)` precision. Repeat run `6b38030f6d9e4ab1a0ee6a4ba0dc5d6e` returned `status=skipped`. + +## 2026-08-07: GBK CSV output + +- All generated CSV files now use GBK without a UTF-8 BOM: the seven full source conversions, the merged summary, repaired history exports, and uploaded history files. `manifest.json`, API JSON, and database character sets remain unchanged. +- The encoding is centralized in `CSV_ENCODING`; tests verify GBK Chinese bytes, absence of the UTF-8 BOM, summary parsing, and history parsing. diff --git a/docs/assumptions.md b/docs/assumptions.md index be75dbd..54a91c6 100644 --- a/docs/assumptions.md +++ b/docs/assumptions.md @@ -8,7 +8,7 @@ - Only `Sheet0` is converted and merged. The `指标(计数器)` sheet is metadata and is not included in the summary. - NR CGI is derived as `{gNBplmn}-{gNBId}-{cellId}`. - 4G CGI is derived as `460-00-{eNodeBID}-{小区ID}`, using the equivalent node and cell column names in each source schema. -- Output CSV files use UTF-8 with BOM so they open correctly in Excel. +- All output CSV files use GBK without a UTF-8 BOM. `manifest.json` remains UTF-8 JSON. - Source files are read-only. The script writes only below its configured output directory. - MySQL is the scheduled-run output. The table uses one `metric_time DATETIME` column containing the source KPI start time. - A successful run transactionally refreshes the selected hour and deletes rows for all other hours. A selected hour older than the newest database hour is rejected to prevent data rollback. diff --git a/main.py b/main.py index 722a864..3bef053 100644 --- a/main.py +++ b/main.py @@ -82,6 +82,7 @@ LTE_PLMN = "460-00" DATABASE_NAME = "interference_etl" DATABASE_TABLE = "interference_hourly_summary" DEFAULT_HISTORY_ROOT = f"{DEFAULT_ROOT}/干扰历史数据" +CSV_ENCODING = "gbk" HEADER_FDD = ( "开始时间", @@ -1082,7 +1083,7 @@ def sql_decimal_literal(value: str, field: str, nullable: bool = False) -> str: def write_csv(path: Path, header: tuple[str, ...], rows: list[tuple[object, ...]]) -> None: - with path.open("w", encoding="utf-8-sig", newline="") as file: + with path.open("w", encoding=CSV_ENCODING, newline="") as file: writer = csv.writer(file) writer.writerow(header) for row in rows: @@ -1098,7 +1099,7 @@ def dict_csv_bytes(header: tuple[str, ...], rows: list[dict[str, str]]) -> bytes writer = csv.DictWriter(output, fieldnames=header, extrasaction="raise") writer.writeheader() writer.writerows(rows) - return output.getvalue().encode("utf-8-sig") + return output.getvalue().encode(CSV_ENCODING) def ensure_scoped(root: Path, target: Path) -> None: diff --git a/tests/test_pipeline.py b/tests/test_pipeline.py index b0bab0c..466fdc1 100644 --- a/tests/test_pipeline.py +++ b/tests/test_pipeline.py @@ -38,7 +38,11 @@ class PipelineTest(unittest.TestCase): self.assertEqual(result.name, TARGET_KEY) converted = sorted((result / "converted").glob("*.csv")) self.assertEqual(len(converted), 7) - with (result / f"interference_summary_{TARGET_KEY}.csv").open(encoding="utf-8-sig", newline="") as file: + summary_path = result / f"interference_summary_{TARGET_KEY}.csv" + summary_bytes = summary_path.read_bytes() + self.assertFalse(summary_bytes.startswith(b"\xef\xbb\xbf")) + self.assertIn("小区".encode("gbk"), summary_bytes) + with summary_path.open(encoding="gbk", newline="") as file: rows = list(csv.DictReader(file)) self.assertEqual(len(rows), 7) self.assertEqual( @@ -94,7 +98,7 @@ class PipelineTest(unittest.TestCase): self.assertEqual(manifest["database"]["metric_time"], "2026-07-31 10:00:00") self.assertEqual(len(store.rows), 7) self.assertEqual(history.path, "/history/2026-07-31/干扰数据处理结果_20260731100000.csv") - archived_rows = list(csv.DictReader(io.StringIO(history.payload.decode("utf-8-sig")))) + archived_rows = list(csv.DictReader(io.StringIO(history.payload.decode("gbk")))) self.assertEqual(list(archived_rows[0]), list(main.SUMMARY_HEADER)) self.assertEqual(len(archived_rows), 7) @@ -321,7 +325,7 @@ class PipelineTest(unittest.TestCase): self.assertIsNone(result) self.assertEqual(store.rows_for_time_calls, 1) self.assertEqual(history.path, "/history/2026-07-31/干扰数据处理结果_20260731100000.csv") - archived_rows = list(csv.DictReader(io.StringIO(history.payload.decode("utf-8-sig")))) + archived_rows = list(csv.DictReader(io.StringIO(history.payload.decode("gbk")))) self.assertEqual(archived_rows, [stored_row]) def test_api_history_store_creates_date_directory_and_uploads_csv(self) -> None: