KIA PDF OCR: 408 pages -> structured JSON (339 sections, 1M chars)
This commit is contained in:
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,178 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""
|
||||||
|
OCR KIA service manual PDF → structured JSON.
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
python3 ocr_kia_pdf.py [--pages N] [--start N]
|
||||||
|
|
||||||
|
Output: kia_data/kia_ocr.json
|
||||||
|
"""
|
||||||
|
|
||||||
|
import pymupdf
|
||||||
|
import subprocess
|
||||||
|
import tempfile
|
||||||
|
import os
|
||||||
|
import json
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
import argparse
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
PDF_PATH = "/mnt/y/Torrents/KIA4_by_kiario.pdf"
|
||||||
|
OUT_DIR = Path(__file__).parent.parent / "kia_data"
|
||||||
|
OUT_JSON = OUT_DIR / "kia_ocr.json"
|
||||||
|
|
||||||
|
# OCR language: rus for Russian, eng for English terms
|
||||||
|
OCR_LANG = "rus+eng"
|
||||||
|
# Zoom factor for rendering (2 = 2x, better OCR quality)
|
||||||
|
ZOOM = 2
|
||||||
|
|
||||||
|
|
||||||
|
def ocr_page(pixmap, page_num=0) -> str:
|
||||||
|
"""Run tesseract on a pixmap, return text. Retry on timeout with explicit cleanup."""
|
||||||
|
# Save pixmap to temp file, close before OCR
|
||||||
|
fd, img_path = tempfile.mkstemp(suffix=".png")
|
||||||
|
os.close(fd)
|
||||||
|
pixmap.save(img_path)
|
||||||
|
|
||||||
|
for attempt in range(3):
|
||||||
|
proc = None
|
||||||
|
try:
|
||||||
|
proc = subprocess.Popen(
|
||||||
|
["tesseract", img_path, "stdout", "-l", OCR_LANG],
|
||||||
|
stdout=subprocess.PIPE,
|
||||||
|
stderr=subprocess.DEVNULL,
|
||||||
|
text=True,
|
||||||
|
)
|
||||||
|
stdout, _ = proc.communicate(timeout=60)
|
||||||
|
if proc.returncode != 0 and proc.returncode is not None:
|
||||||
|
raise RuntimeError(f"tesseract exit {proc.returncode}")
|
||||||
|
os.unlink(img_path)
|
||||||
|
return stdout.strip()
|
||||||
|
except subprocess.TimeoutExpired:
|
||||||
|
if proc:
|
||||||
|
proc.kill()
|
||||||
|
try:
|
||||||
|
proc.wait(timeout=5)
|
||||||
|
except subprocess.TimeoutExpired:
|
||||||
|
pass
|
||||||
|
if attempt < 2:
|
||||||
|
print(f" p{page_num}: timeout, retry {attempt+2}/3...", flush=True)
|
||||||
|
else:
|
||||||
|
print(f" p{page_num}: SKIP (timeout after 3 retries)", flush=True)
|
||||||
|
os.unlink(img_path)
|
||||||
|
return ""
|
||||||
|
except Exception as e:
|
||||||
|
if proc:
|
||||||
|
proc.kill()
|
||||||
|
try:
|
||||||
|
proc.wait(timeout=5)
|
||||||
|
except subprocess.TimeoutExpired:
|
||||||
|
pass
|
||||||
|
os.unlink(img_path)
|
||||||
|
print(f" p{page_num}: ERROR {e}", flush=True)
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
|
def is_section_header(line: str) -> bool:
|
||||||
|
"""
|
||||||
|
Heuristic: detect section/chapter headers.
|
||||||
|
Examples:
|
||||||
|
- "ГЛАВА 1. ОБЩИЕ СВЕДЕНИЯ"
|
||||||
|
- "1. ТЕХНИЧЕСКОЕ ОБСЛУЖИВАНИЕ"
|
||||||
|
- "ДВИГАТЕЛЬ"
|
||||||
|
"""
|
||||||
|
line = line.strip()
|
||||||
|
if not line or len(line) < 3:
|
||||||
|
return False
|
||||||
|
|
||||||
|
# Numbered chapter: "1.", "1.2", "ГЛАВА 1"
|
||||||
|
if re.match(r"^(ГЛАВА|РАЗДЕЛ|ЧАСТЬ)\s+\d+", line, re.IGNORECASE):
|
||||||
|
return True
|
||||||
|
if re.match(r"^\d+\.\s+[А-ЯA-Z]", line):
|
||||||
|
return True
|
||||||
|
if re.match(r"^\d+\.\d+\.?\s+[А-ЯA-Z]", line):
|
||||||
|
return True
|
||||||
|
|
||||||
|
# ALL CAPS, 3+ words, reasonable length (not a full sentence)
|
||||||
|
if line.isupper() and len(line.split()) >= 2 and 5 < len(line) < 100:
|
||||||
|
return True
|
||||||
|
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def parse_pdf(start_page=0, max_pages=None):
|
||||||
|
"""OCR all pages and structure into sections."""
|
||||||
|
doc = pymupdf.open(PDF_PATH)
|
||||||
|
total_pages = doc.page_count
|
||||||
|
end_page = min(total_pages, start_page + max_pages) if max_pages else total_pages
|
||||||
|
|
||||||
|
print(f"PDF: {total_pages} pages, processing {start_page}–{end_page-1}", flush=True)
|
||||||
|
|
||||||
|
sections = []
|
||||||
|
current_section = {"title": "Начало", "pages": [], "text": ""}
|
||||||
|
mat = pymupdf.Matrix(ZOOM, ZOOM)
|
||||||
|
|
||||||
|
for i in range(start_page, end_page):
|
||||||
|
page = doc[i]
|
||||||
|
pix = page.get_pixmap(matrix=mat)
|
||||||
|
text = ocr_page(pix, page_num=i)
|
||||||
|
|
||||||
|
if not text:
|
||||||
|
print(f" p{i}: [empty]")
|
||||||
|
continue
|
||||||
|
|
||||||
|
# Try to detect section header on this page
|
||||||
|
lines = text.split("\n")
|
||||||
|
header_found = False
|
||||||
|
for line in lines[:5]: # check first 5 lines
|
||||||
|
if is_section_header(line):
|
||||||
|
# Save previous section
|
||||||
|
if current_section["text"].strip():
|
||||||
|
sections.append(current_section)
|
||||||
|
current_section = {"title": line.strip(), "pages": [], "text": ""}
|
||||||
|
header_found = True
|
||||||
|
break
|
||||||
|
|
||||||
|
current_section["pages"].append(i)
|
||||||
|
current_section["text"] += text + "\n"
|
||||||
|
|
||||||
|
if (i - start_page) % 20 == 0:
|
||||||
|
print(f" p{i}: {len(text)} chars, sections so far: {len(sections)}", flush=True)
|
||||||
|
|
||||||
|
# Save last section
|
||||||
|
if current_section["text"].strip():
|
||||||
|
sections.append(current_section)
|
||||||
|
|
||||||
|
doc.close()
|
||||||
|
return {
|
||||||
|
"source": PDF_PATH,
|
||||||
|
"total_pages": total_pages,
|
||||||
|
"pages_processed": end_page - start_page,
|
||||||
|
"sections": sections,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
parser = argparse.ArgumentParser(description="OCR KIA PDF → JSON")
|
||||||
|
parser.add_argument("--pages", type=int, default=None, help="Max pages to process")
|
||||||
|
parser.add_argument("--start", type=int, default=0, help="Start page")
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
OUT_DIR.mkdir(parents=True, exist_ok=True)
|
||||||
|
|
||||||
|
data = parse_pdf(start_page=args.start, max_pages=args.pages)
|
||||||
|
|
||||||
|
print(f"\nSections: {len(data['sections'])}")
|
||||||
|
total_chars = sum(len(s["text"]) for s in data["sections"])
|
||||||
|
print(f"Total chars: {total_chars:,}")
|
||||||
|
|
||||||
|
with open(OUT_JSON, "w", encoding="utf-8") as f:
|
||||||
|
json.dump(data, f, ensure_ascii=False, indent=2)
|
||||||
|
|
||||||
|
print(f"Saved: {OUT_JSON} ({os.path.getsize(OUT_JSON)/1024/1024:.1f} MB)")
|
||||||
|
print("Done!")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
Reference in New Issue
Block a user