diff --git a/deploy/drhider/drhider.py b/deploy/drhider/drhider.py index abcbe06..5fcaec7 100644 --- a/deploy/drhider/drhider.py +++ b/deploy/drhider/drhider.py @@ -252,8 +252,9 @@ class TwoPassObfuscator: self._sorted_keys.clear() def _convert_pdfs_to_docx(self, files: List[Tuple[str, bytes, str]]) -> List[Tuple[str, bytes, str]]: - """Конвертировать PDF в DOCX через LibreOffice. При совпадении имён — _из_pdf.""" - import subprocess, tempfile + """Конвертировать PDF в DOCX через pdfplumber. При совпадении имён — _из_pdf.""" + import pdfplumber + from docx import Document as DocxDocument result = [] existing_names = {f[0] for f in files} for fname, content, ctype in files: @@ -261,34 +262,36 @@ class TwoPassObfuscator: result.append((fname, content, ctype)) continue try: - with tempfile.NamedTemporaryFile(suffix='.pdf', delete=False) as tmp: - tmp.write(content) - pdf_path = tmp.name - outdir = tempfile.mkdtemp() - subprocess.run(['libreoffice', '--headless', '--convert-to', 'docx', - '--outdir', outdir, pdf_path], timeout=30, capture_output=True) - docx_files = [f for f in os.listdir(outdir) if f.endswith('.docx')] - if docx_files: - with open(os.path.join(outdir, docx_files[0]), 'rb') as f: - docx_content = f.read() - new_name = fname[:-4] + '.docx' - if new_name in existing_names: - new_name = fname[:-4] + '_из_pdf.docx' - existing_names.add(new_name) - result.append((new_name, docx_content, ctype)) - else: - log.warning("PDF→DOCX failed for %s", fname) - result.append((fname, content, ctype)) + doc = DocxDocument() + with pdfplumber.open(io.BytesIO(content)) as pdf: + for page in pdf.pages: + tables = page.extract_tables() + for table in tables: + if table: + rows = [[str(c or "").strip() for c in (row or [])] for row in table] + rows = [r for r in rows if any(r)] + if rows: + t = doc.add_table(rows=len(rows), cols=len(rows[0])) + t.style = 'Table Grid' + for ri, row in enumerate(rows): + for ci, cell_text in enumerate(row): + t.rows[ri].cells[ci].text = cell_text + text = page.extract_text() + if text: + for line in text.split('\n'): + line = line.strip() + if line: + doc.add_paragraph(line) + buf = io.BytesIO() + doc.save(buf) + new_name = fname[:-4] + '.docx' + if new_name in existing_names: + new_name = fname[:-4] + '_из_pdf.docx' + existing_names.add(new_name) + result.append((new_name, buf.getvalue(), ctype)) except Exception as e: log.warning("PDF→DOCX error for %s: %s", fname, e) result.append((fname, content, ctype)) - finally: - if os.path.exists(pdf_path): - os.unlink(pdf_path) - if os.path.exists(outdir): - for f in os.listdir(outdir): - os.unlink(os.path.join(outdir, f)) - os.rmdir(outdir) return result def _expand_zips(self, files: List[Tuple[str, bytes, str]]) -> List[Tuple[str, bytes, str]]: diff --git a/site/services/drhider.py b/site/services/drhider.py index ebfea7a..3428e64 100644 --- a/site/services/drhider.py +++ b/site/services/drhider.py @@ -252,8 +252,9 @@ class TwoPassObfuscator: self._sorted_keys.clear() def _convert_pdfs_to_docx(self, files: List[Tuple[str, bytes, str]]) -> List[Tuple[str, bytes, str]]: - """Конвертировать PDF в DOCX через LibreOffice. При совпадении имён — _из_pdf.""" - import subprocess, tempfile + """Конвертировать PDF в DOCX через pdfplumber. При совпадении имён — _из_pdf.""" + import pdfplumber + from docx import Document as DocxDocument result = [] existing_names = {f[0] for f in files} for fname, content, ctype in files: @@ -261,34 +262,36 @@ class TwoPassObfuscator: result.append((fname, content, ctype)) continue try: - with tempfile.NamedTemporaryFile(suffix='.pdf', delete=False) as tmp: - tmp.write(content) - pdf_path = tmp.name - outdir = tempfile.mkdtemp() - subprocess.run(['libreoffice', '--headless', '--convert-to', 'docx', - '--outdir', outdir, pdf_path], timeout=30, capture_output=True) - docx_files = [f for f in os.listdir(outdir) if f.endswith('.docx')] - if docx_files: - with open(os.path.join(outdir, docx_files[0]), 'rb') as f: - docx_content = f.read() - new_name = fname[:-4] + '.docx' - if new_name in existing_names: - new_name = fname[:-4] + '_из_pdf.docx' - existing_names.add(new_name) - result.append((new_name, docx_content, ctype)) - else: - log.warning("PDF→DOCX failed for %s", fname) - result.append((fname, content, ctype)) + doc = DocxDocument() + with pdfplumber.open(io.BytesIO(content)) as pdf: + for page in pdf.pages: + tables = page.extract_tables() + for table in tables: + if table: + rows = [[str(c or "").strip() for c in (row or [])] for row in table] + rows = [r for r in rows if any(r)] + if rows: + t = doc.add_table(rows=len(rows), cols=len(rows[0])) + t.style = 'Table Grid' + for ri, row in enumerate(rows): + for ci, cell_text in enumerate(row): + t.rows[ri].cells[ci].text = cell_text + text = page.extract_text() + if text: + for line in text.split('\n'): + line = line.strip() + if line: + doc.add_paragraph(line) + buf = io.BytesIO() + doc.save(buf) + new_name = fname[:-4] + '.docx' + if new_name in existing_names: + new_name = fname[:-4] + '_из_pdf.docx' + existing_names.add(new_name) + result.append((new_name, buf.getvalue(), ctype)) except Exception as e: log.warning("PDF→DOCX error for %s: %s", fname, e) result.append((fname, content, ctype)) - finally: - if os.path.exists(pdf_path): - os.unlink(pdf_path) - if os.path.exists(outdir): - for f in os.listdir(outdir): - os.unlink(os.path.join(outdir, f)) - os.rmdir(outdir) return result def _expand_zips(self, files: List[Tuple[str, bytes, str]]) -> List[Tuple[str, bytes, str]]: diff --git a/site/templates/drhider.html b/site/templates/drhider.html index 5695ec4..c372ab7 100644 --- a/site/templates/drhider.html +++ b/site/templates/drhider.html @@ -84,7 +84,7 @@ Сверка договоров | DrHider - v1.11 + v1.12