KIA PDF OCR: 408 pages -> structured JSON (339 sections, 1M chars)
This commit is contained in:
@@ -0,0 +1,178 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
OCR KIA service manual PDF → structured JSON.
|
||||
|
||||
Usage:
|
||||
python3 ocr_kia_pdf.py [--pages N] [--start N]
|
||||
|
||||
Output: kia_data/kia_ocr.json
|
||||
"""
|
||||
|
||||
import pymupdf
|
||||
import subprocess
|
||||
import tempfile
|
||||
import os
|
||||
import json
|
||||
import re
|
||||
import sys
|
||||
import argparse
|
||||
from pathlib import Path
|
||||
|
||||
PDF_PATH = "/mnt/y/Torrents/KIA4_by_kiario.pdf"
|
||||
OUT_DIR = Path(__file__).parent.parent / "kia_data"
|
||||
OUT_JSON = OUT_DIR / "kia_ocr.json"
|
||||
|
||||
# OCR language: rus for Russian, eng for English terms
|
||||
OCR_LANG = "rus+eng"
|
||||
# Zoom factor for rendering (2 = 2x, better OCR quality)
|
||||
ZOOM = 2
|
||||
|
||||
|
||||
def ocr_page(pixmap, page_num=0) -> str:
|
||||
"""Run tesseract on a pixmap, return text. Retry on timeout with explicit cleanup."""
|
||||
# Save pixmap to temp file, close before OCR
|
||||
fd, img_path = tempfile.mkstemp(suffix=".png")
|
||||
os.close(fd)
|
||||
pixmap.save(img_path)
|
||||
|
||||
for attempt in range(3):
|
||||
proc = None
|
||||
try:
|
||||
proc = subprocess.Popen(
|
||||
["tesseract", img_path, "stdout", "-l", OCR_LANG],
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.DEVNULL,
|
||||
text=True,
|
||||
)
|
||||
stdout, _ = proc.communicate(timeout=60)
|
||||
if proc.returncode != 0 and proc.returncode is not None:
|
||||
raise RuntimeError(f"tesseract exit {proc.returncode}")
|
||||
os.unlink(img_path)
|
||||
return stdout.strip()
|
||||
except subprocess.TimeoutExpired:
|
||||
if proc:
|
||||
proc.kill()
|
||||
try:
|
||||
proc.wait(timeout=5)
|
||||
except subprocess.TimeoutExpired:
|
||||
pass
|
||||
if attempt < 2:
|
||||
print(f" p{page_num}: timeout, retry {attempt+2}/3...", flush=True)
|
||||
else:
|
||||
print(f" p{page_num}: SKIP (timeout after 3 retries)", flush=True)
|
||||
os.unlink(img_path)
|
||||
return ""
|
||||
except Exception as e:
|
||||
if proc:
|
||||
proc.kill()
|
||||
try:
|
||||
proc.wait(timeout=5)
|
||||
except subprocess.TimeoutExpired:
|
||||
pass
|
||||
os.unlink(img_path)
|
||||
print(f" p{page_num}: ERROR {e}", flush=True)
|
||||
return ""
|
||||
|
||||
|
||||
def is_section_header(line: str) -> bool:
|
||||
"""
|
||||
Heuristic: detect section/chapter headers.
|
||||
Examples:
|
||||
- "ГЛАВА 1. ОБЩИЕ СВЕДЕНИЯ"
|
||||
- "1. ТЕХНИЧЕСКОЕ ОБСЛУЖИВАНИЕ"
|
||||
- "ДВИГАТЕЛЬ"
|
||||
"""
|
||||
line = line.strip()
|
||||
if not line or len(line) < 3:
|
||||
return False
|
||||
|
||||
# Numbered chapter: "1.", "1.2", "ГЛАВА 1"
|
||||
if re.match(r"^(ГЛАВА|РАЗДЕЛ|ЧАСТЬ)\s+\d+", line, re.IGNORECASE):
|
||||
return True
|
||||
if re.match(r"^\d+\.\s+[А-ЯA-Z]", line):
|
||||
return True
|
||||
if re.match(r"^\d+\.\d+\.?\s+[А-ЯA-Z]", line):
|
||||
return True
|
||||
|
||||
# ALL CAPS, 3+ words, reasonable length (not a full sentence)
|
||||
if line.isupper() and len(line.split()) >= 2 and 5 < len(line) < 100:
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
|
||||
def parse_pdf(start_page=0, max_pages=None):
|
||||
"""OCR all pages and structure into sections."""
|
||||
doc = pymupdf.open(PDF_PATH)
|
||||
total_pages = doc.page_count
|
||||
end_page = min(total_pages, start_page + max_pages) if max_pages else total_pages
|
||||
|
||||
print(f"PDF: {total_pages} pages, processing {start_page}–{end_page-1}", flush=True)
|
||||
|
||||
sections = []
|
||||
current_section = {"title": "Начало", "pages": [], "text": ""}
|
||||
mat = pymupdf.Matrix(ZOOM, ZOOM)
|
||||
|
||||
for i in range(start_page, end_page):
|
||||
page = doc[i]
|
||||
pix = page.get_pixmap(matrix=mat)
|
||||
text = ocr_page(pix, page_num=i)
|
||||
|
||||
if not text:
|
||||
print(f" p{i}: [empty]")
|
||||
continue
|
||||
|
||||
# Try to detect section header on this page
|
||||
lines = text.split("\n")
|
||||
header_found = False
|
||||
for line in lines[:5]: # check first 5 lines
|
||||
if is_section_header(line):
|
||||
# Save previous section
|
||||
if current_section["text"].strip():
|
||||
sections.append(current_section)
|
||||
current_section = {"title": line.strip(), "pages": [], "text": ""}
|
||||
header_found = True
|
||||
break
|
||||
|
||||
current_section["pages"].append(i)
|
||||
current_section["text"] += text + "\n"
|
||||
|
||||
if (i - start_page) % 20 == 0:
|
||||
print(f" p{i}: {len(text)} chars, sections so far: {len(sections)}", flush=True)
|
||||
|
||||
# Save last section
|
||||
if current_section["text"].strip():
|
||||
sections.append(current_section)
|
||||
|
||||
doc.close()
|
||||
return {
|
||||
"source": PDF_PATH,
|
||||
"total_pages": total_pages,
|
||||
"pages_processed": end_page - start_page,
|
||||
"sections": sections,
|
||||
}
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="OCR KIA PDF → JSON")
|
||||
parser.add_argument("--pages", type=int, default=None, help="Max pages to process")
|
||||
parser.add_argument("--start", type=int, default=0, help="Start page")
|
||||
args = parser.parse_args()
|
||||
|
||||
OUT_DIR.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
data = parse_pdf(start_page=args.start, max_pages=args.pages)
|
||||
|
||||
print(f"\nSections: {len(data['sections'])}")
|
||||
total_chars = sum(len(s["text"]) for s in data["sections"])
|
||||
print(f"Total chars: {total_chars:,}")
|
||||
|
||||
with open(OUT_JSON, "w", encoding="utf-8") as f:
|
||||
json.dump(data, f, ensure_ascii=False, indent=2)
|
||||
|
||||
print(f"Saved: {OUT_JSON} ({os.path.getsize(OUT_JSON)/1024/1024:.1f} MB)")
|
||||
print("Done!")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user