152 lines
6.0 KiB
Python
152 lines
6.0 KiB
Python
"""Upload blueprint — загрузка, конвертация, распаковка."""
|
|
import io, os, base64, hashlib, zipfile
|
|
import httpx
|
|
from flask import Blueprint, request, jsonify, send_file
|
|
from services.parse import parse_file
|
|
from db import documents
|
|
import config
|
|
|
|
upload_bp = Blueprint("upload", __name__)
|
|
|
|
ALLOWED = {"pdf", "docx", "doc", "zip"}
|
|
|
|
|
|
def _check_ext(filename: str) -> str | None:
|
|
ext = filename.rsplit(".", 1)[-1].lower() if "." in filename else ""
|
|
if ext not in ALLOWED:
|
|
return f"unsupported format: .{ext} (allowed: {', '.join(sorted(ALLOWED))})"
|
|
return None
|
|
|
|
|
|
def _unzip(data: bytes):
|
|
"""Распаковать ZIP → (ok, files, error)."""
|
|
MAX_FILES = 500
|
|
MAX_UNCOMPRESSED = 500 * 1024 * 1024 # 500 MB
|
|
files = []
|
|
total = 0
|
|
try:
|
|
with zipfile.ZipFile(io.BytesIO(data)) as zf:
|
|
if len(zf.namelist()) > MAX_FILES:
|
|
return False, None, f"too many files in ZIP (max {MAX_FILES})"
|
|
for info in zf.infolist():
|
|
if info.is_dir():
|
|
continue
|
|
name = os.path.basename(info.filename)
|
|
if not name or ".." in name or "/" in name or "\\" in name:
|
|
continue
|
|
raw = zf.read(info)
|
|
total += len(raw)
|
|
if total > MAX_UNCOMPRESSED:
|
|
return False, None, "total uncompressed size exceeds 500 MB"
|
|
ext = name.rsplit(".", 1)[-1].lower() if "." in name else ""
|
|
files.append({
|
|
"filename": name,
|
|
"ext": ext,
|
|
"size": len(raw),
|
|
"data_b64": base64.b64encode(raw).decode(),
|
|
})
|
|
except zipfile.BadZipFile:
|
|
return False, None, "invalid ZIP archive"
|
|
return True, files, None
|
|
|
|
|
|
def _convert(filename: str, data: bytes) -> bytes:
|
|
""".doc → .docx через внешний libreoffice-сервис. Возвращает docx-байты."""
|
|
try:
|
|
resp = httpx.post(
|
|
config.CONVERT_SERVICE_URL + "/convert",
|
|
files={"file": (filename, data, "application/msword")},
|
|
timeout=120,
|
|
)
|
|
except httpx.TimeoutException:
|
|
raise Exception("conversion timeout")
|
|
if resp.status_code != 200:
|
|
try:
|
|
err = resp.json().get("error", "conversion failed")
|
|
except Exception:
|
|
err = "conversion failed"
|
|
raise Exception(err)
|
|
return resp.content
|
|
|
|
|
|
def _store_and_parse(filename: str, data: bytes, batch_id, contract_id, zip_source=None, mime_type="application/octet-stream"):
|
|
"""Общая логика: дедуп → insert в БД → авто-парсинг. Возвращает dict-результат."""
|
|
content_hash = hashlib.sha256(data).hexdigest()[:16]
|
|
|
|
# Дедупликация по хешу
|
|
if batch_id:
|
|
existing = documents.get_by_hash(batch_id, content_hash)
|
|
if existing:
|
|
return {"ok": False, "error": "duplicate", "doc_id": existing["id"], "duplicate_of": True}
|
|
|
|
doc = documents.insert(
|
|
filename=filename,
|
|
mime_type=mime_type,
|
|
original_bytes=base64.b64encode(data).decode(),
|
|
batch_id=batch_id,
|
|
zip_source=zip_source,
|
|
content_hash=content_hash,
|
|
)
|
|
|
|
# Авто-парсинг
|
|
try:
|
|
result = parse_file(filename, data)
|
|
if result["status"] == "parsed":
|
|
documents.set_parsed(doc["id"], result["elements"])
|
|
parsed = {"status": "parsed", "element_count": result.get("element_count", 0)}
|
|
else:
|
|
documents.set_error(doc["id"], result.get("error", "parse failed"))
|
|
parsed = {"status": "error", "error": result.get("error", "parse failed")}
|
|
except Exception as e:
|
|
documents.set_error(doc["id"], str(e))
|
|
parsed = {"status": "error", "error": str(e)}
|
|
|
|
return {"ok": True, "doc_id": doc["id"], "contract_id": contract_id, "parsed": parsed}
|
|
|
|
|
|
@upload_bp.route("/upload", methods=["POST"])
|
|
def upload():
|
|
"""Загрузка одного файла + авто-парсинг → БД (прямой multipart)."""
|
|
f = request.files.get("files")
|
|
if not f:
|
|
return jsonify(ok=False, error="no file"), 400
|
|
|
|
err = _check_ext(f.filename)
|
|
if err:
|
|
return jsonify(ok=False, error=err), 400
|
|
|
|
data = f.read()
|
|
batch_id = request.form.get("batch_id")
|
|
zip_source = request.form.get("zip_source")
|
|
contract_id = request.form.get("contract_id")
|
|
|
|
result = _store_and_parse(f.filename, data, batch_id, contract_id, zip_source, f.content_type or "application/octet-stream")
|
|
if not result["ok"]:
|
|
return jsonify(ok=result["ok"], error=result.get("error"), doc_id=result.get("doc_id")), 200
|
|
return jsonify(ok=True, doc_id=result["doc_id"], contract_id=result["contract_id"], parsed=result["parsed"])
|
|
|
|
|
|
def contracts_upload_sink(name, content, batch_id=None, contract_id=None, zip_source=None):
|
|
"""Sink для переиспользуемого модуля upload: вставить файл в documents + авто-парсинг.
|
|
|
|
Вызывается модулем create_upload_refs_blueprint(cfg, sink=...) на каждый
|
|
вытянутый с ВМ файл. .doc конвертируется в .docx через внешний LibreOffice
|
|
(парсер умеет только .docx). Возвращает dict ok/doc_id/contract_id/parsed.
|
|
"""
|
|
if name.lower().endswith(".doc"):
|
|
content = _convert(name, content)
|
|
name = name[:-4] + ".docx"
|
|
return _store_and_parse(name, content, batch_id, contract_id, zip_source)
|
|
|
|
|
|
@upload_bp.route("/unzip-upload", methods=["POST"])
|
|
def unzip_upload():
|
|
"""Распаковать ZIP → список файлов (base64 для фронтенда, прямой multipart)."""
|
|
f = request.files.get("files")
|
|
if not f:
|
|
return jsonify(ok=False, error="no file"), 400
|
|
ok, files, err = _unzip(f.read())
|
|
if not ok:
|
|
return jsonify(ok=False, error=err), 400
|
|
return jsonify(ok=True, files=files)
|