v0.0.41: время на конкретный файл в колонке Статус (бэк считает, без общего LLM)
Deploy drhider / validate (push) Canceled after 0s

This commit is contained in:
“Naeel”
2026-08-19 13:46:00 +04:00
parent 020f9d43c6
commit 0453827b8b
5 changed files with 61 additions and 12 deletions
+11 -5
View File
@@ -9,6 +9,7 @@
import io
import os
import time
import logging
from typing import Dict, List, Tuple, Callable, Optional
@@ -94,9 +95,9 @@ class TwoPassObfuscator:
Args:
files: [(filename, content_bytes, content_type), ...]
progress_cb: Опциональный коллбек (phase, idx, total, fname),
где phase ∈ {"start", "done"}, idx — 0-based индекс.
Вызывается вокруг финальной обработки каждого файла.
progress_cb: Опциональный коллбек (phase, idx, total, fname, elapsed),
где phase ∈ {"start", "done"}, idx — 0-based индекс,
elapsed — время обработки конкретного файла (сек).
Returns:
(zip_bytes, csv_string):
@@ -112,18 +113,21 @@ class TwoPassObfuscator:
# Извлекаем Markdown из каждого файла
all_texts: Dict[str, str] = {}
total = len(files)
file_times: List[float] = [0.0] * total
for i, (fname, content, ctype) in enumerate(files):
display_name = os.path.basename(fname) or fname
if progress_cb:
progress_cb("start", i, total, display_name)
progress_cb("start", i, total, display_name, 0.0)
t0 = time.time()
text = extractor.extract_text(fname, content, ctype)
all_texts[fname] = text
# Regex-сканирование (быстрое, локальное)
if text and not text.startswith("[DOC binary"):
scanner.scan_regex(text, self._mapping, self._counters)
file_times[i] += time.time() - t0
# LLM-сканирование (получает уже найденное regex'ом чтобы не дублировать)
if self._llm_client:
@@ -141,6 +145,7 @@ class TwoPassObfuscator:
fname = in_fname
display_name = os.path.basename(in_fname) or in_fname
t0 = time.time()
obf_content = content # По умолчанию — без изменений
if fname.endswith('.doc'):
@@ -163,9 +168,10 @@ class TwoPassObfuscator:
obf_content = replaced.encode('utf-8')
fname = md_name
file_times[i] += time.time() - t0
results.append((fname, obf_content))
if progress_cb:
progress_cb("done", i, total, display_name)
progress_cb("done", i, total, display_name, round(file_times[i], 2))
# ── Сборка результата ──
csv_str = builder.build_mapping_csv(self._mapping)