from __future__ import annotations import json import os from pathlib import Path import re from http.server import SimpleHTTPRequestHandler, ThreadingHTTPServer from tempfile import NamedTemporaryFile from urllib.parse import urlparse from bank_importer.parser import StatementParseError, parse_statement ROOT = Path(__file__).resolve().parent WEB_ROOT = ROOT / "web" MAX_UPLOAD_BYTES = 20 * 1024 * 1024 class AppHandler(SimpleHTTPRequestHandler): def __init__(self, *args, **kwargs): super().__init__(*args, directory=str(WEB_ROOT), **kwargs) def do_POST(self) -> None: if urlparse(self.path).path != "/api/parse": self.send_error(404) return try: filename, content = self._read_uploaded_file() suffix = Path(filename).suffix.lower() if suffix not in {".xls", ".xlsx"}: raise ValueError("仅支持 .xls 或 .xlsx 银行流水文件。") with NamedTemporaryFile(suffix=suffix, delete=False) as temp_file: temp_path = Path(temp_file.name) temp_file.write(content) try: batches = parse_statement(temp_path) except StatementParseError as exc: message = str(exc).replace(temp_path.name, filename) raise StatementParseError(message) from exc finally: temp_path.unlink(missing_ok=True) batch = batches[0] self._send_json( 200, { "status": "parsed", "bank": batch.bank_name, "template": batch.template_id, "header_row": batch.header_row, "period_start": batch.period_start.isoformat() if batch.period_start else None, "period_end": batch.period_end.isoformat() if batch.period_end else None, "transactions": len(batch.transactions), "warnings": list(batch.warnings), }, ) except (ValueError, StatementParseError) as exc: self._send_json(422, {"status": "exception", "message": str(exc)}) except Exception: self._send_json( 500, {"status": "error", "message": "文件解析失败,请检查文件是否完整。"}, ) def _read_uploaded_file(self) -> tuple[str, bytes]: content_length = int(self.headers.get("Content-Length", "0")) if content_length <= 0 or content_length > MAX_UPLOAD_BYTES: raise ValueError("文件为空或超过 20 MB 限制。") content_type = self.headers.get("Content-Type", "") boundary_match = re.search(r"boundary=(?:\"([^\"]+)\"|([^;]+))", content_type) if not boundary_match: raise ValueError("上传请求缺少文件边界。") boundary = (boundary_match.group(1) or boundary_match.group(2)).encode() body = self.rfile.read(content_length) for part in body.split(b"--" + boundary): if b'name="file"' not in part: continue header, separator, content = part.partition(b"\r\n\r\n") if not separator: continue filename_match = re.search(br'filename="([^\"]+)"', header) filename = ( filename_match.group(1).decode("utf-8", errors="replace") if filename_match else "statement.xlsx" ) return os.path.basename(filename), content.rstrip(b"\r\n") raise ValueError("上传请求中没有找到文件。") def _send_json(self, status: int, payload: dict[str, object]) -> None: content = json.dumps(payload, ensure_ascii=False).encode("utf-8") self.send_response(status) self.send_header("Content-Type", "application/json; charset=utf-8") self.send_header("Content-Length", str(len(content))) self.send_header("Cache-Control", "no-store") self.end_headers() self.wfile.write(content) def main() -> None: host = os.environ.get("APP_HOST", "127.0.0.1") port = int(os.environ.get("APP_PORT", "4173")) server = ThreadingHTTPServer((host, port), AppHandler) print(f"Serving on http://{host}:{port}") server.serve_forever() if __name__ == "__main__": main()