feat: 支持历史原始数据下载

This commit is contained in:
2026-05-19 16:45:10 +08:00
parent b25847dbb8
commit 332ed7055d
4 changed files with 159 additions and 5 deletions
+69
View File
@@ -1,14 +1,60 @@
import zipfile
from contextlib import suppress
from datetime import datetime
from pathlib import Path
from fastapi import APIRouter, Body, HTTPException
from fastapi.responses import FileResponse
from starlette.background import BackgroundTask
from app import state
from app.config import CACHE_DIR
from app.utils.files import format_size, get_dir_size
router = APIRouter(tags=["history"])
def _remove_file(path: Path) -> None:
with suppress(OSError):
path.unlink()
def _safe_filename_part(value: str) -> str:
safe = "".join(char if char.isalnum() or char in {"-", "_"} else "_" for char in value)
return safe.strip("_") or "history"
def _get_safe_work_dir(record_id: str) -> tuple[Path, str]:
record = state.history_manager.get(record_id)
if not record:
raise HTTPException(status_code=404, detail="记录不存在")
if record.status in {"pending", "processing"}:
raise HTTPException(status_code=409, detail="任务尚未完成,暂不能下载历史数据")
work_dir = Path(record.work_dir).resolve()
cache_dir = CACHE_DIR.resolve()
try:
work_dir.relative_to(cache_dir)
except ValueError as exc:
raise HTTPException(status_code=400, detail="历史目录不合法") from exc
if not work_dir.exists() or not work_dir.is_dir():
raise HTTPException(status_code=404, detail="历史数据目录不存在")
return work_dir, record.id
def _zip_directory(source_dir: Path, archive_path: Path) -> None:
with zipfile.ZipFile(archive_path, "w", compression=zipfile.ZIP_DEFLATED, allowZip64=True) as archive:
for item in source_dir.rglob("*"):
arcname = item.relative_to(source_dir).as_posix()
if item.is_dir():
archive.writestr(f"{arcname}/", "")
elif item.is_file():
archive.write(item, arcname)
@router.post("/api/history")
async def get_history(limit: int = Body(50, embed=True)):
return {"records": state.history_manager.list(limit)}
@@ -51,3 +97,26 @@ async def get_history_size(record_id: str = Body(..., embed=True)):
size = get_dir_size(work_dir)
return {"success": True, "size": size, "size_formatted": format_size(size)}
@router.post("/api/history/download")
async def download_history(record_id: str = Body(..., embed=True)):
work_dir, safe_record_id = _get_safe_work_dir(record_id)
export_dir = CACHE_DIR / ".downloads"
export_dir.mkdir(parents=True, exist_ok=True)
timestamp = datetime.now().strftime("%Y%m%d_%H%M%S")
filename = f"{_safe_filename_part(safe_record_id)}_{timestamp}.zip"
archive_path = export_dir / filename
try:
_zip_directory(work_dir, archive_path)
except Exception:
_remove_file(archive_path)
raise
return FileResponse(
path=str(archive_path),
filename=filename,
media_type="application/zip",
background=BackgroundTask(_remove_file, archive_path),
)
+54 -3
View File
@@ -11,6 +11,7 @@ import zipfile
import multiprocessing
import pandas as pd
from pathlib import Path
from threading import Lock
from typing import Any, Callable, Dict, Generator, List, Optional, Tuple
from datetime import datetime
from concurrent.futures import ThreadPoolExecutor, as_completed
@@ -114,6 +115,8 @@ class DataProcessor:
self.db = DatabaseManager(config)
self.results: Dict[str, Any] = {}
self._explicit_type_fields: set[str] = set()
self._generated_csv_files: set[Path] = set()
self._generated_csv_lock = Lock()
# 预编译字段映射,避免重复查找
self._field_map, self._type_map = self._build_field_map()
@@ -266,24 +269,70 @@ class DataProcessor:
# 优先尝试 UTF-8(现代 ZIP 文件标准)
try:
with zipfile.ZipFile(zip_file, 'r', metadata_encoding='utf-8') as zf:
zf.extractall(zip_file.parent)
self._extract_zip_members(zf, zip_file.parent)
return
except (UnicodeDecodeError, zipfile.BadZipFile):
# UTF-8 失败,尝试 GBK(Windows 中文系统常用)
try:
with zipfile.ZipFile(zip_file, 'r', metadata_encoding='gbk') as zf:
zf.extractall(zip_file.parent)
self._extract_zip_members(zf, zip_file.parent)
return
except (UnicodeDecodeError, zipfile.BadZipFile):
# GBK 也失败,尝试 CP437(DOS 编码)
try:
with zipfile.ZipFile(zip_file, 'r', metadata_encoding='cp437') as zf:
zf.extractall(zip_file.parent)
self._extract_zip_members(zf, zip_file.parent)
return
except Exception as e:
# 所有编码都失败
raise Exception(f"无法解压 ZIP 文件,编码检测失败: {e}")
def _extract_zip_members(self, zf: zipfile.ZipFile, target_dir: Path) -> None:
root = target_dir.resolve()
for member in zf.infolist():
member_name = member.filename.replace("\\", "/")
target_path = (root / member_name).resolve()
try:
target_path.relative_to(root)
except ValueError:
self.logger.warning(f"跳过不安全的 ZIP 条目: {member.filename}")
continue
if member.is_dir():
target_path.mkdir(parents=True, exist_ok=True)
continue
target_path.parent.mkdir(parents=True, exist_ok=True)
with zf.open(member) as source, target_path.open("wb") as target:
shutil.copyfileobj(source, target)
if target_path.suffix.lower() == ".csv":
self._remember_generated_csv(target_path)
def _remember_generated_csv(self, csv_file: Path) -> None:
with self._generated_csv_lock:
self._generated_csv_files.add(csv_file.resolve())
def _delete_generated_csv(self, csv_file: Path) -> None:
csv_path = csv_file.resolve()
with self._generated_csv_lock:
if csv_path not in self._generated_csv_files:
return
self._generated_csv_files.remove(csv_path)
try:
csv_path.relative_to(self.work_dir.resolve())
except ValueError:
return
try:
if csv_path.exists() and csv_path.is_file():
csv_path.unlink()
self.logger.info(f"已清理临时 CSV: {csv_path.relative_to(self.work_dir)}")
except OSError as exc:
self.logger.warning(f"清理临时 CSV 失败 {csv_path}: {exc}")
def _scan_files(self, directory: Path, extensions: List[str]) -> Generator[Path, None, None]:
"""扫描指定扩展名的文件"""
for ext in extensions:
@@ -305,6 +354,7 @@ class DataProcessor:
# 直接读取并写入,不做额外处理
df = xl.parse(sheet_name)
df.to_csv(output_file, index=False, encoding='utf-8')
self._remember_generated_csv(output_file)
processed += 1
xl.close()
@@ -759,6 +809,7 @@ class DataProcessor:
csv_file, table_name, conn, table_created
)
total_rows += rows
self._delete_generated_csv(csv_file)
# 每处理 10 个文件报告一次进度
if i % 10 == 0:
+6
View File
@@ -1,4 +1,10 @@
# 项目上下文记录
## 2026-05-19:清理生成 CSV 并支持历史原始数据下载
- `DataProcessor` 现在会追踪 ZIP 解压出的 CSV 和 Excel 转换生成的 CSV,只有这些处理过程中生成的临时 CSV 会在对应 CSV 成功导入后自动删除;原始 ZIP、Excel 和用户本来上传/远程下载得到的原始 CSV 不会被误删。
- ZIP 解压从 `extractall()` 改为逐条安全解压,会跳过越界路径条目,并在解压 CSV 时登记为后续可清理的临时文件。
- `POST /api/history/download` 会校验历史任务目录必须位于 `cache/` 下,任务完成后才能下载;接口将整个历史工作目录压缩为 ZIP 返回,并通过 `BackgroundTask` 在响应结束后删除临时压缩包。
- `frontend/src/components/HistoryPanel.vue` 在历史列表的“详情”左侧增加“下载”按钮,下载时显示 loading,未完成任务禁用下载,避免重复点击和下载不完整的历史数据。
- 已执行 `.venv\Scripts\python.exe -m compileall app`、`uvx --offline ruff check .` 和 `npm run build`,均通过;本次未启动浏览器或 headless Chrome。
## 2026-05-19:调整数值异常值归零和完成后日志高度
+30 -2
View File
@@ -18,6 +18,16 @@
</div>
<span class="record-size">{{ statusText(record.status) }}</span>
<div class="history-actions">
<n-button
size="small"
tertiary
:loading="downloadingRecordId === record.id"
:disabled="record.status === 'pending' || record.status === 'processing'"
@click="downloadHistory(record)"
>
<template #icon><n-icon><DownloadOutline /></n-icon></template>
下载
</n-button>
<n-button size="small" tertiary @click="openDetail(record.id)">详情</n-button>
<n-button size="small" tertiary type="error" @click="confirmDelete(record.id)">删除</n-button>
</div>
@@ -86,9 +96,9 @@
<script setup lang="ts">
import { computed, onBeforeUnmount, onMounted, ref } from 'vue';
import { useDialog, useMessage } from 'naive-ui';
import { CopyOutline, FileTrayOutline, RefreshOutline, TrashOutline } from '@vicons/ionicons5';
import { CopyOutline, DownloadOutline, FileTrayOutline, RefreshOutline, TrashOutline } from '@vicons/ionicons5';
import { apiGet, apiPost } from '../api/client';
import { apiGet, apiPost, download as downloadFile } from '../api/client';
import type { CacheSize, HistoryDetail, HistoryRecord } from '../types';
import { resetPageHeader, setPageHeader } from '../composables/pageHeader';
@@ -103,6 +113,7 @@ const cacheSize = ref<CacheSize | null>(null);
const recordSize = ref('-');
const recordSizeLoading = ref(false);
const loading = ref(false);
const downloadingRecordId = ref<string | null>(null);
let recordSizeToken = 0;
const totalSizeText = computed(() => `总占用: ${cacheSize.value?.size_formatted || '计算中...'}`);
@@ -165,6 +176,23 @@ async function openDetail(recordId: string) {
}
}
async function downloadHistory(record: HistoryRecord) {
if (record.status === 'pending' || record.status === 'processing') {
message.warning('任务尚未完成,暂不能下载历史数据');
return;
}
downloadingRecordId.value = record.id;
try {
await downloadFile('/api/history/download', { record_id: record.id }, `${record.id}.zip`);
message.success('历史数据下载已开始');
} catch (error) {
message.error(error instanceof Error ? error.message : '下载历史数据失败');
} finally {
downloadingRecordId.value = null;
}
}
async function loadRecordSize(recordId: string) {
const currentToken = ++recordSizeToken;
recordSize.value = '计算中...';