Deploy contracts-flask / validate (push) Canceled after 0s
- /api/upload_refs: pull файлов с ВМ (SSRF-guard, лимит 50МБ, ретраи, DELETE) - uploadFile: PUT на WebDAV /contracts-upload/ → refs → pull - nginx: location /contracts-upload/ (WebDAV + CORS для managed-фронта)
227 lines
8.5 KiB
Python
227 lines
8.5 KiB
Python
"""Upload blueprint — загрузка, конвертация, распаковка."""
|
||
import io, os, time, base64, hashlib, zipfile
|
||
import httpx
|
||
from flask import Blueprint, request, jsonify, send_file
|
||
from services.parse import parse_file
|
||
from db import documents
|
||
import config
|
||
|
||
upload_bp = Blueprint("upload", __name__)
|
||
|
||
ALLOWED = {"pdf", "docx", "doc", "zip"}
|
||
|
||
|
||
def _check_ext(filename: str) -> str | None:
|
||
ext = filename.rsplit(".", 1)[-1].lower() if "." in filename else ""
|
||
if ext not in ALLOWED:
|
||
return f"unsupported format: .{ext} (allowed: {', '.join(sorted(ALLOWED))})"
|
||
return None
|
||
|
||
|
||
def _safe_name(name: str) -> str:
|
||
"""Санитизация имени файла: только basename, защита от path traversal."""
|
||
name = (name or "").replace("\\", "/").rsplit("/", 1)[-1].strip()
|
||
if not name or name in (".", ".."):
|
||
return "file.bin"
|
||
return name[:255]
|
||
|
||
|
||
def _pull_with_retries(url: str, timeout: int = 120) -> bytes:
|
||
"""Скачать файл с ВМ-буфера с ретраями (egress, без лимита шлюза)."""
|
||
last: Exception | None = None
|
||
for attempt in range(1, config.PULL_RETRIES + 1):
|
||
try:
|
||
resp = httpx.get(url, timeout=timeout, follow_redirects=False)
|
||
if resp.status_code == 200:
|
||
return resp.content
|
||
last = Exception(f"HTTP {resp.status_code}")
|
||
except Exception as e:
|
||
last = e
|
||
if attempt < config.PULL_RETRIES:
|
||
time.sleep(config.PULL_RETRY_DELAY)
|
||
raise last or Exception("pull failed")
|
||
|
||
|
||
def _delete_from_vm(url: str) -> None:
|
||
"""Best-effort удаление файла с ВМ-буфера."""
|
||
try:
|
||
httpx.delete(url, timeout=30)
|
||
except Exception:
|
||
pass
|
||
|
||
|
||
def _store_and_parse(filename: str, data: bytes, batch_id, contract_id, zip_source=None, mime_type="application/octet-stream"):
|
||
"""Общая логика: дедуп → insert в БД → авто-парсинг. Возвращает dict-результат."""
|
||
content_hash = hashlib.sha256(data).hexdigest()[:16]
|
||
|
||
# Дедупликация по хешу
|
||
if batch_id:
|
||
existing = documents.get_by_hash(batch_id, content_hash)
|
||
if existing:
|
||
return {"ok": False, "error": "duplicate", "doc_id": existing["id"], "duplicate_of": True}
|
||
|
||
doc = documents.insert(
|
||
filename=filename,
|
||
mime_type=mime_type,
|
||
original_bytes=base64.b64encode(data).decode(),
|
||
batch_id=batch_id,
|
||
zip_source=zip_source,
|
||
content_hash=content_hash,
|
||
)
|
||
|
||
# Авто-парсинг
|
||
try:
|
||
result = parse_file(filename, data)
|
||
if result["status"] == "parsed":
|
||
documents.set_parsed(doc["id"], result["elements"])
|
||
parsed = {"status": "parsed", "element_count": result.get("element_count", 0)}
|
||
else:
|
||
documents.set_error(doc["id"], result.get("error", "parse failed"))
|
||
parsed = {"status": "error", "error": result.get("error", "parse failed")}
|
||
except Exception as e:
|
||
documents.set_error(doc["id"], str(e))
|
||
parsed = {"status": "error", "error": str(e)}
|
||
|
||
return {"ok": True, "doc_id": doc["id"], "contract_id": contract_id, "parsed": parsed}
|
||
|
||
|
||
@upload_bp.route("/upload", methods=["POST"])
|
||
def upload():
|
||
"""Загрузка одного файла + авто-парсинг → БД (прямой multipart)."""
|
||
f = request.files.get("files")
|
||
if not f:
|
||
return jsonify(ok=False, error="no file"), 400
|
||
|
||
err = _check_ext(f.filename)
|
||
if err:
|
||
return jsonify(ok=False, error=err), 400
|
||
|
||
data = f.read()
|
||
batch_id = request.form.get("batch_id")
|
||
zip_source = request.form.get("zip_source")
|
||
contract_id = request.form.get("contract_id")
|
||
|
||
result = _store_and_parse(f.filename, data, batch_id, contract_id, zip_source, f.content_type or "application/octet-stream")
|
||
if not result["ok"]:
|
||
return jsonify(ok=result["ok"], error=result.get("error"), doc_id=result.get("doc_id")), 200
|
||
return jsonify(ok=True, doc_id=result["doc_id"], contract_id=result["contract_id"], parsed=result["parsed"])
|
||
|
||
|
||
@upload_bp.route("/api/upload_refs", methods=["POST"])
|
||
def upload_refs():
|
||
"""Загрузка через ВМ-буфер (паттерн drhider): бэк тянет файлы с ВМ.
|
||
|
||
Тело (маленькое, <64КБ): {batch_id, contract_id, zip_source, files:[{name,size,url}]}.
|
||
Для каждой ссылки: SSRF-проверка → лимит → pull с ретраями → store+parse → DELETE с ВМ.
|
||
"""
|
||
data = request.get_json(silent=True) or {}
|
||
files = data.get("files") or []
|
||
batch_id = data.get("batch_id")
|
||
contract_id = data.get("contract_id")
|
||
zip_source = data.get("zip_source")
|
||
|
||
if not files:
|
||
return jsonify(ok=False, error="no files"), 400
|
||
|
||
results = []
|
||
for ref in files:
|
||
name = _safe_name(str(ref.get("name", "") or ""))
|
||
size = int(ref.get("size") or 0)
|
||
url = str(ref.get("url", "") or "")
|
||
|
||
# SSRF-защита: тянуть можно ТОЛЬКО с доверенного ВМ-буфера
|
||
if not url.startswith(config.VM_UPLOAD_PREFIX):
|
||
results.append({"name": name, "ok": False, "error": "invalid url (SSRF guard)"})
|
||
continue
|
||
|
||
# Лимит по заявленному размеру
|
||
if size > config.VM_UPLOAD_MAX_BYTES:
|
||
_delete_from_vm(url)
|
||
results.append({"name": name, "ok": False, "error": f"file too large: {size} bytes (max {config.VM_UPLOAD_MAX_BYTES})", "skipped": True})
|
||
continue
|
||
|
||
try:
|
||
content = _pull_with_retries(url)
|
||
except Exception as e:
|
||
results.append({"name": name, "ok": False, "error": f"pull failed: {e}"})
|
||
continue
|
||
|
||
# Реальная проверка размера после pull
|
||
if len(content) > config.VM_UPLOAD_MAX_BYTES:
|
||
_delete_from_vm(url)
|
||
results.append({"name": name, "ok": False, "error": "file too large after pull", "skipped": True})
|
||
continue
|
||
|
||
_delete_from_vm(url)
|
||
stored = _store_and_parse(name, content, batch_id, contract_id, zip_source)
|
||
results.append({"name": name, **stored})
|
||
|
||
return jsonify(ok=True, results=results)
|
||
|
||
|
||
@upload_bp.route("/convert-doc", methods=["POST"])
|
||
def convert_doc():
|
||
""".doc → .docx через внешний libreoffice-сервис (HTTP)."""
|
||
f = request.files.get("files")
|
||
if not f:
|
||
return jsonify(ok=False, error="no file"), 400
|
||
|
||
try:
|
||
resp = httpx.post(
|
||
config.CONVERT_SERVICE_URL + "/convert",
|
||
files={"file": (f.filename, f.read(), f.content_type or "application/msword")},
|
||
timeout=120,
|
||
)
|
||
if resp.status_code == 200:
|
||
return send_file(
|
||
io.BytesIO(resp.content),
|
||
mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
||
)
|
||
data = resp.json()
|
||
return jsonify(ok=False, error=data.get("error", "conversion failed")), 500
|
||
except httpx.TimeoutException:
|
||
return jsonify(ok=False, error="conversion timeout"), 500
|
||
except Exception as e:
|
||
return jsonify(ok=False, error=str(e)), 500
|
||
|
||
|
||
@upload_bp.route("/unzip-upload", methods=["POST"])
|
||
def unzip_upload():
|
||
"""Распаковать ZIP → список файлов (base64 для фронтенда)."""
|
||
f = request.files.get("files")
|
||
if not f:
|
||
return jsonify(ok=False, error="no file"), 400
|
||
|
||
data = f.read()
|
||
MAX_FILES = 500
|
||
MAX_UNCOMPRESSED = 500 * 1024 * 1024 # 500 MB
|
||
|
||
files = []
|
||
total = 0
|
||
|
||
with zipfile.ZipFile(io.BytesIO(data)) as zf:
|
||
if len(zf.namelist()) > MAX_FILES:
|
||
return jsonify(ok=False, error=f"too many files in ZIP (max {MAX_FILES})"), 400
|
||
|
||
for info in zf.infolist():
|
||
if info.is_dir():
|
||
continue
|
||
name = os.path.basename(info.filename)
|
||
if not name or ".." in name or "/" in name or "\\" in name:
|
||
continue
|
||
|
||
raw = zf.read(info)
|
||
total += len(raw)
|
||
if total > MAX_UNCOMPRESSED:
|
||
return jsonify(ok=False, error="total uncompressed size exceeds 500 MB"), 400
|
||
|
||
ext = name.rsplit(".", 1)[-1].lower() if "." in name else ""
|
||
files.append({
|
||
"filename": name,
|
||
"ext": ext,
|
||
"size": len(raw),
|
||
"data_b64": base64.b64encode(raw).decode(),
|
||
})
|
||
|
||
return jsonify(ok=True, files=files)
|