feat: 支持历史原始数据下载
This commit is contained in:
@@ -1,14 +1,60 @@
|
|||||||
|
import zipfile
|
||||||
|
from contextlib import suppress
|
||||||
|
from datetime import datetime
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
from fastapi import APIRouter, Body, HTTPException
|
from fastapi import APIRouter, Body, HTTPException
|
||||||
|
from fastapi.responses import FileResponse
|
||||||
|
from starlette.background import BackgroundTask
|
||||||
|
|
||||||
from app import state
|
from app import state
|
||||||
|
from app.config import CACHE_DIR
|
||||||
from app.utils.files import format_size, get_dir_size
|
from app.utils.files import format_size, get_dir_size
|
||||||
|
|
||||||
|
|
||||||
router = APIRouter(tags=["history"])
|
router = APIRouter(tags=["history"])
|
||||||
|
|
||||||
|
|
||||||
|
def _remove_file(path: Path) -> None:
|
||||||
|
with suppress(OSError):
|
||||||
|
path.unlink()
|
||||||
|
|
||||||
|
|
||||||
|
def _safe_filename_part(value: str) -> str:
|
||||||
|
safe = "".join(char if char.isalnum() or char in {"-", "_"} else "_" for char in value)
|
||||||
|
return safe.strip("_") or "history"
|
||||||
|
|
||||||
|
|
||||||
|
def _get_safe_work_dir(record_id: str) -> tuple[Path, str]:
|
||||||
|
record = state.history_manager.get(record_id)
|
||||||
|
if not record:
|
||||||
|
raise HTTPException(status_code=404, detail="记录不存在")
|
||||||
|
if record.status in {"pending", "processing"}:
|
||||||
|
raise HTTPException(status_code=409, detail="任务尚未完成,暂不能下载历史数据")
|
||||||
|
|
||||||
|
work_dir = Path(record.work_dir).resolve()
|
||||||
|
cache_dir = CACHE_DIR.resolve()
|
||||||
|
try:
|
||||||
|
work_dir.relative_to(cache_dir)
|
||||||
|
except ValueError as exc:
|
||||||
|
raise HTTPException(status_code=400, detail="历史目录不合法") from exc
|
||||||
|
|
||||||
|
if not work_dir.exists() or not work_dir.is_dir():
|
||||||
|
raise HTTPException(status_code=404, detail="历史数据目录不存在")
|
||||||
|
|
||||||
|
return work_dir, record.id
|
||||||
|
|
||||||
|
|
||||||
|
def _zip_directory(source_dir: Path, archive_path: Path) -> None:
|
||||||
|
with zipfile.ZipFile(archive_path, "w", compression=zipfile.ZIP_DEFLATED, allowZip64=True) as archive:
|
||||||
|
for item in source_dir.rglob("*"):
|
||||||
|
arcname = item.relative_to(source_dir).as_posix()
|
||||||
|
if item.is_dir():
|
||||||
|
archive.writestr(f"{arcname}/", "")
|
||||||
|
elif item.is_file():
|
||||||
|
archive.write(item, arcname)
|
||||||
|
|
||||||
|
|
||||||
@router.post("/api/history")
|
@router.post("/api/history")
|
||||||
async def get_history(limit: int = Body(50, embed=True)):
|
async def get_history(limit: int = Body(50, embed=True)):
|
||||||
return {"records": state.history_manager.list(limit)}
|
return {"records": state.history_manager.list(limit)}
|
||||||
@@ -51,3 +97,26 @@ async def get_history_size(record_id: str = Body(..., embed=True)):
|
|||||||
size = get_dir_size(work_dir)
|
size = get_dir_size(work_dir)
|
||||||
return {"success": True, "size": size, "size_formatted": format_size(size)}
|
return {"success": True, "size": size, "size_formatted": format_size(size)}
|
||||||
|
|
||||||
|
|
||||||
|
@router.post("/api/history/download")
|
||||||
|
async def download_history(record_id: str = Body(..., embed=True)):
|
||||||
|
work_dir, safe_record_id = _get_safe_work_dir(record_id)
|
||||||
|
export_dir = CACHE_DIR / ".downloads"
|
||||||
|
export_dir.mkdir(parents=True, exist_ok=True)
|
||||||
|
|
||||||
|
timestamp = datetime.now().strftime("%Y%m%d_%H%M%S")
|
||||||
|
filename = f"{_safe_filename_part(safe_record_id)}_{timestamp}.zip"
|
||||||
|
archive_path = export_dir / filename
|
||||||
|
|
||||||
|
try:
|
||||||
|
_zip_directory(work_dir, archive_path)
|
||||||
|
except Exception:
|
||||||
|
_remove_file(archive_path)
|
||||||
|
raise
|
||||||
|
|
||||||
|
return FileResponse(
|
||||||
|
path=str(archive_path),
|
||||||
|
filename=filename,
|
||||||
|
media_type="application/zip",
|
||||||
|
background=BackgroundTask(_remove_file, archive_path),
|
||||||
|
)
|
||||||
|
|||||||
+54
-3
@@ -11,6 +11,7 @@ import zipfile
|
|||||||
import multiprocessing
|
import multiprocessing
|
||||||
import pandas as pd
|
import pandas as pd
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
from threading import Lock
|
||||||
from typing import Any, Callable, Dict, Generator, List, Optional, Tuple
|
from typing import Any, Callable, Dict, Generator, List, Optional, Tuple
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||||
@@ -114,6 +115,8 @@ class DataProcessor:
|
|||||||
self.db = DatabaseManager(config)
|
self.db = DatabaseManager(config)
|
||||||
self.results: Dict[str, Any] = {}
|
self.results: Dict[str, Any] = {}
|
||||||
self._explicit_type_fields: set[str] = set()
|
self._explicit_type_fields: set[str] = set()
|
||||||
|
self._generated_csv_files: set[Path] = set()
|
||||||
|
self._generated_csv_lock = Lock()
|
||||||
|
|
||||||
# 预编译字段映射,避免重复查找
|
# 预编译字段映射,避免重复查找
|
||||||
self._field_map, self._type_map = self._build_field_map()
|
self._field_map, self._type_map = self._build_field_map()
|
||||||
@@ -266,24 +269,70 @@ class DataProcessor:
|
|||||||
# 优先尝试 UTF-8(现代 ZIP 文件标准)
|
# 优先尝试 UTF-8(现代 ZIP 文件标准)
|
||||||
try:
|
try:
|
||||||
with zipfile.ZipFile(zip_file, 'r', metadata_encoding='utf-8') as zf:
|
with zipfile.ZipFile(zip_file, 'r', metadata_encoding='utf-8') as zf:
|
||||||
zf.extractall(zip_file.parent)
|
self._extract_zip_members(zf, zip_file.parent)
|
||||||
return
|
return
|
||||||
except (UnicodeDecodeError, zipfile.BadZipFile):
|
except (UnicodeDecodeError, zipfile.BadZipFile):
|
||||||
# UTF-8 失败,尝试 GBK(Windows 中文系统常用)
|
# UTF-8 失败,尝试 GBK(Windows 中文系统常用)
|
||||||
try:
|
try:
|
||||||
with zipfile.ZipFile(zip_file, 'r', metadata_encoding='gbk') as zf:
|
with zipfile.ZipFile(zip_file, 'r', metadata_encoding='gbk') as zf:
|
||||||
zf.extractall(zip_file.parent)
|
self._extract_zip_members(zf, zip_file.parent)
|
||||||
return
|
return
|
||||||
except (UnicodeDecodeError, zipfile.BadZipFile):
|
except (UnicodeDecodeError, zipfile.BadZipFile):
|
||||||
# GBK 也失败,尝试 CP437(DOS 编码)
|
# GBK 也失败,尝试 CP437(DOS 编码)
|
||||||
try:
|
try:
|
||||||
with zipfile.ZipFile(zip_file, 'r', metadata_encoding='cp437') as zf:
|
with zipfile.ZipFile(zip_file, 'r', metadata_encoding='cp437') as zf:
|
||||||
zf.extractall(zip_file.parent)
|
self._extract_zip_members(zf, zip_file.parent)
|
||||||
return
|
return
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
# 所有编码都失败
|
# 所有编码都失败
|
||||||
raise Exception(f"无法解压 ZIP 文件,编码检测失败: {e}")
|
raise Exception(f"无法解压 ZIP 文件,编码检测失败: {e}")
|
||||||
|
|
||||||
|
def _extract_zip_members(self, zf: zipfile.ZipFile, target_dir: Path) -> None:
|
||||||
|
root = target_dir.resolve()
|
||||||
|
for member in zf.infolist():
|
||||||
|
member_name = member.filename.replace("\\", "/")
|
||||||
|
target_path = (root / member_name).resolve()
|
||||||
|
|
||||||
|
try:
|
||||||
|
target_path.relative_to(root)
|
||||||
|
except ValueError:
|
||||||
|
self.logger.warning(f"跳过不安全的 ZIP 条目: {member.filename}")
|
||||||
|
continue
|
||||||
|
|
||||||
|
if member.is_dir():
|
||||||
|
target_path.mkdir(parents=True, exist_ok=True)
|
||||||
|
continue
|
||||||
|
|
||||||
|
target_path.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
with zf.open(member) as source, target_path.open("wb") as target:
|
||||||
|
shutil.copyfileobj(source, target)
|
||||||
|
|
||||||
|
if target_path.suffix.lower() == ".csv":
|
||||||
|
self._remember_generated_csv(target_path)
|
||||||
|
|
||||||
|
def _remember_generated_csv(self, csv_file: Path) -> None:
|
||||||
|
with self._generated_csv_lock:
|
||||||
|
self._generated_csv_files.add(csv_file.resolve())
|
||||||
|
|
||||||
|
def _delete_generated_csv(self, csv_file: Path) -> None:
|
||||||
|
csv_path = csv_file.resolve()
|
||||||
|
with self._generated_csv_lock:
|
||||||
|
if csv_path not in self._generated_csv_files:
|
||||||
|
return
|
||||||
|
self._generated_csv_files.remove(csv_path)
|
||||||
|
|
||||||
|
try:
|
||||||
|
csv_path.relative_to(self.work_dir.resolve())
|
||||||
|
except ValueError:
|
||||||
|
return
|
||||||
|
|
||||||
|
try:
|
||||||
|
if csv_path.exists() and csv_path.is_file():
|
||||||
|
csv_path.unlink()
|
||||||
|
self.logger.info(f"已清理临时 CSV: {csv_path.relative_to(self.work_dir)}")
|
||||||
|
except OSError as exc:
|
||||||
|
self.logger.warning(f"清理临时 CSV 失败 {csv_path}: {exc}")
|
||||||
|
|
||||||
def _scan_files(self, directory: Path, extensions: List[str]) -> Generator[Path, None, None]:
|
def _scan_files(self, directory: Path, extensions: List[str]) -> Generator[Path, None, None]:
|
||||||
"""扫描指定扩展名的文件"""
|
"""扫描指定扩展名的文件"""
|
||||||
for ext in extensions:
|
for ext in extensions:
|
||||||
@@ -305,6 +354,7 @@ class DataProcessor:
|
|||||||
# 直接读取并写入,不做额外处理
|
# 直接读取并写入,不做额外处理
|
||||||
df = xl.parse(sheet_name)
|
df = xl.parse(sheet_name)
|
||||||
df.to_csv(output_file, index=False, encoding='utf-8')
|
df.to_csv(output_file, index=False, encoding='utf-8')
|
||||||
|
self._remember_generated_csv(output_file)
|
||||||
processed += 1
|
processed += 1
|
||||||
|
|
||||||
xl.close()
|
xl.close()
|
||||||
@@ -759,6 +809,7 @@ class DataProcessor:
|
|||||||
csv_file, table_name, conn, table_created
|
csv_file, table_name, conn, table_created
|
||||||
)
|
)
|
||||||
total_rows += rows
|
total_rows += rows
|
||||||
|
self._delete_generated_csv(csv_file)
|
||||||
|
|
||||||
# 每处理 10 个文件报告一次进度
|
# 每处理 10 个文件报告一次进度
|
||||||
if i % 10 == 0:
|
if i % 10 == 0:
|
||||||
|
|||||||
@@ -1,4 +1,10 @@
|
|||||||
# 项目上下文记录
|
# 项目上下文记录
|
||||||
|
## 2026-05-19:清理生成 CSV 并支持历史原始数据下载
|
||||||
|
- `DataProcessor` 现在会追踪 ZIP 解压出的 CSV 和 Excel 转换生成的 CSV,只有这些处理过程中生成的临时 CSV 会在对应 CSV 成功导入后自动删除;原始 ZIP、Excel 和用户本来上传/远程下载得到的原始 CSV 不会被误删。
|
||||||
|
- ZIP 解压从 `extractall()` 改为逐条安全解压,会跳过越界路径条目,并在解压 CSV 时登记为后续可清理的临时文件。
|
||||||
|
- `POST /api/history/download` 会校验历史任务目录必须位于 `cache/` 下,任务完成后才能下载;接口将整个历史工作目录压缩为 ZIP 返回,并通过 `BackgroundTask` 在响应结束后删除临时压缩包。
|
||||||
|
- `frontend/src/components/HistoryPanel.vue` 在历史列表的“详情”左侧增加“下载”按钮,下载时显示 loading,未完成任务禁用下载,避免重复点击和下载不完整的历史数据。
|
||||||
|
- 已执行 `.venv\Scripts\python.exe -m compileall app`、`uvx --offline ruff check .` 和 `npm run build`,均通过;本次未启动浏览器或 headless Chrome。
|
||||||
|
|
||||||
## 2026-05-19:调整数值异常值归零和完成后日志高度
|
## 2026-05-19:调整数值异常值归零和完成后日志高度
|
||||||
|
|
||||||
|
|||||||
@@ -18,6 +18,16 @@
|
|||||||
</div>
|
</div>
|
||||||
<span class="record-size">{{ statusText(record.status) }}</span>
|
<span class="record-size">{{ statusText(record.status) }}</span>
|
||||||
<div class="history-actions">
|
<div class="history-actions">
|
||||||
|
<n-button
|
||||||
|
size="small"
|
||||||
|
tertiary
|
||||||
|
:loading="downloadingRecordId === record.id"
|
||||||
|
:disabled="record.status === 'pending' || record.status === 'processing'"
|
||||||
|
@click="downloadHistory(record)"
|
||||||
|
>
|
||||||
|
<template #icon><n-icon><DownloadOutline /></n-icon></template>
|
||||||
|
下载
|
||||||
|
</n-button>
|
||||||
<n-button size="small" tertiary @click="openDetail(record.id)">详情</n-button>
|
<n-button size="small" tertiary @click="openDetail(record.id)">详情</n-button>
|
||||||
<n-button size="small" tertiary type="error" @click="confirmDelete(record.id)">删除</n-button>
|
<n-button size="small" tertiary type="error" @click="confirmDelete(record.id)">删除</n-button>
|
||||||
</div>
|
</div>
|
||||||
@@ -86,9 +96,9 @@
|
|||||||
<script setup lang="ts">
|
<script setup lang="ts">
|
||||||
import { computed, onBeforeUnmount, onMounted, ref } from 'vue';
|
import { computed, onBeforeUnmount, onMounted, ref } from 'vue';
|
||||||
import { useDialog, useMessage } from 'naive-ui';
|
import { useDialog, useMessage } from 'naive-ui';
|
||||||
import { CopyOutline, FileTrayOutline, RefreshOutline, TrashOutline } from '@vicons/ionicons5';
|
import { CopyOutline, DownloadOutline, FileTrayOutline, RefreshOutline, TrashOutline } from '@vicons/ionicons5';
|
||||||
|
|
||||||
import { apiGet, apiPost } from '../api/client';
|
import { apiGet, apiPost, download as downloadFile } from '../api/client';
|
||||||
import type { CacheSize, HistoryDetail, HistoryRecord } from '../types';
|
import type { CacheSize, HistoryDetail, HistoryRecord } from '../types';
|
||||||
import { resetPageHeader, setPageHeader } from '../composables/pageHeader';
|
import { resetPageHeader, setPageHeader } from '../composables/pageHeader';
|
||||||
|
|
||||||
@@ -103,6 +113,7 @@ const cacheSize = ref<CacheSize | null>(null);
|
|||||||
const recordSize = ref('-');
|
const recordSize = ref('-');
|
||||||
const recordSizeLoading = ref(false);
|
const recordSizeLoading = ref(false);
|
||||||
const loading = ref(false);
|
const loading = ref(false);
|
||||||
|
const downloadingRecordId = ref<string | null>(null);
|
||||||
let recordSizeToken = 0;
|
let recordSizeToken = 0;
|
||||||
|
|
||||||
const totalSizeText = computed(() => `总占用: ${cacheSize.value?.size_formatted || '计算中...'}`);
|
const totalSizeText = computed(() => `总占用: ${cacheSize.value?.size_formatted || '计算中...'}`);
|
||||||
@@ -165,6 +176,23 @@ async function openDetail(recordId: string) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async function downloadHistory(record: HistoryRecord) {
|
||||||
|
if (record.status === 'pending' || record.status === 'processing') {
|
||||||
|
message.warning('任务尚未完成,暂不能下载历史数据');
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
downloadingRecordId.value = record.id;
|
||||||
|
try {
|
||||||
|
await downloadFile('/api/history/download', { record_id: record.id }, `${record.id}.zip`);
|
||||||
|
message.success('历史数据下载已开始');
|
||||||
|
} catch (error) {
|
||||||
|
message.error(error instanceof Error ? error.message : '下载历史数据失败');
|
||||||
|
} finally {
|
||||||
|
downloadingRecordId.value = null;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
async function loadRecordSize(recordId: string) {
|
async function loadRecordSize(recordId: string) {
|
||||||
const currentToken = ++recordSizeToken;
|
const currentToken = ++recordSizeToken;
|
||||||
recordSize.value = '计算中...';
|
recordSize.value = '计算中...';
|
||||||
|
|||||||
Reference in New Issue
Block a user