KIA PDF OCR: 408 pages -> structured JSON (339 sections, 1M chars)

This commit is contained in:
2026-06-09 08:39:43 +04:00
parent ecb96761e7
commit 6ab90aa312
2 changed files with 2626 additions and 0 deletions
+178
View File
@@ -0,0 +1,178 @@
#!/usr/bin/env python3
"""
OCR KIA service manual PDF → structured JSON.
Usage:
python3 ocr_kia_pdf.py [--pages N] [--start N]
Output: kia_data/kia_ocr.json
"""
import pymupdf
import subprocess
import tempfile
import os
import json
import re
import sys
import argparse
from pathlib import Path
PDF_PATH = "/mnt/y/Torrents/KIA4_by_kiario.pdf"
OUT_DIR = Path(__file__).parent.parent / "kia_data"
OUT_JSON = OUT_DIR / "kia_ocr.json"
# OCR language: rus for Russian, eng for English terms
OCR_LANG = "rus+eng"
# Zoom factor for rendering (2 = 2x, better OCR quality)
ZOOM = 2
def ocr_page(pixmap, page_num=0) -> str:
"""Run tesseract on a pixmap, return text. Retry on timeout with explicit cleanup."""
# Save pixmap to temp file, close before OCR
fd, img_path = tempfile.mkstemp(suffix=".png")
os.close(fd)
pixmap.save(img_path)
for attempt in range(3):
proc = None
try:
proc = subprocess.Popen(
["tesseract", img_path, "stdout", "-l", OCR_LANG],
stdout=subprocess.PIPE,
stderr=subprocess.DEVNULL,
text=True,
)
stdout, _ = proc.communicate(timeout=60)
if proc.returncode != 0 and proc.returncode is not None:
raise RuntimeError(f"tesseract exit {proc.returncode}")
os.unlink(img_path)
return stdout.strip()
except subprocess.TimeoutExpired:
if proc:
proc.kill()
try:
proc.wait(timeout=5)
except subprocess.TimeoutExpired:
pass
if attempt < 2:
print(f" p{page_num}: timeout, retry {attempt+2}/3...", flush=True)
else:
print(f" p{page_num}: SKIP (timeout after 3 retries)", flush=True)
os.unlink(img_path)
return ""
except Exception as e:
if proc:
proc.kill()
try:
proc.wait(timeout=5)
except subprocess.TimeoutExpired:
pass
os.unlink(img_path)
print(f" p{page_num}: ERROR {e}", flush=True)
return ""
def is_section_header(line: str) -> bool:
"""
Heuristic: detect section/chapter headers.
Examples:
- "ГЛАВА 1. ОБЩИЕ СВЕДЕНИЯ"
- "1. ТЕХНИЧЕСКОЕ ОБСЛУЖИВАНИЕ"
- "ДВИГАТЕЛЬ"
"""
line = line.strip()
if not line or len(line) < 3:
return False
# Numbered chapter: "1.", "1.2", "ГЛАВА 1"
if re.match(r"^(ГЛАВА|РАЗДЕЛ|ЧАСТЬ)\s+\d+", line, re.IGNORECASE):
return True
if re.match(r"^\d+\.\s+[А-ЯA-Z]", line):
return True
if re.match(r"^\d+\.\d+\.?\s+[А-ЯA-Z]", line):
return True
# ALL CAPS, 3+ words, reasonable length (not a full sentence)
if line.isupper() and len(line.split()) >= 2 and 5 < len(line) < 100:
return True
return False
def parse_pdf(start_page=0, max_pages=None):
"""OCR all pages and structure into sections."""
doc = pymupdf.open(PDF_PATH)
total_pages = doc.page_count
end_page = min(total_pages, start_page + max_pages) if max_pages else total_pages
print(f"PDF: {total_pages} pages, processing {start_page}{end_page-1}", flush=True)
sections = []
current_section = {"title": "Начало", "pages": [], "text": ""}
mat = pymupdf.Matrix(ZOOM, ZOOM)
for i in range(start_page, end_page):
page = doc[i]
pix = page.get_pixmap(matrix=mat)
text = ocr_page(pix, page_num=i)
if not text:
print(f" p{i}: [empty]")
continue
# Try to detect section header on this page
lines = text.split("\n")
header_found = False
for line in lines[:5]: # check first 5 lines
if is_section_header(line):
# Save previous section
if current_section["text"].strip():
sections.append(current_section)
current_section = {"title": line.strip(), "pages": [], "text": ""}
header_found = True
break
current_section["pages"].append(i)
current_section["text"] += text + "\n"
if (i - start_page) % 20 == 0:
print(f" p{i}: {len(text)} chars, sections so far: {len(sections)}", flush=True)
# Save last section
if current_section["text"].strip():
sections.append(current_section)
doc.close()
return {
"source": PDF_PATH,
"total_pages": total_pages,
"pages_processed": end_page - start_page,
"sections": sections,
}
def main():
parser = argparse.ArgumentParser(description="OCR KIA PDF → JSON")
parser.add_argument("--pages", type=int, default=None, help="Max pages to process")
parser.add_argument("--start", type=int, default=0, help="Start page")
args = parser.parse_args()
OUT_DIR.mkdir(parents=True, exist_ok=True)
data = parse_pdf(start_page=args.start, max_pages=args.pages)
print(f"\nSections: {len(data['sections'])}")
total_chars = sum(len(s["text"]) for s in data["sections"])
print(f"Total chars: {total_chars:,}")
with open(OUT_JSON, "w", encoding="utf-8") as f:
json.dump(data, f, ensure_ascii=False, indent=2)
print(f"Saved: {OUT_JSON} ({os.path.getsize(OUT_JSON)/1024/1024:.1f} MB)")
print("Done!")
if __name__ == "__main__":
main()