54 lines
1.9 KiB
Python
54 lines
1.9 KiB
Python
"""PP-OCRv5 Latin inference using preloaded local model directories only."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import os
|
|
from pathlib import Path
|
|
|
|
|
|
ROOT = Path(os.environ.get("OCR_MODEL_HOME", Path(__file__).resolve().parents[1] / ".cache")).expanduser().resolve()
|
|
MODELS = ROOT / "official_models"
|
|
|
|
|
|
def main() -> None:
|
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
parser.add_argument("image", type=Path)
|
|
parser.add_argument("output", type=Path)
|
|
parser.add_argument("--fine-text", action="store_true")
|
|
args = parser.parse_args()
|
|
image, output = args.image, args.output
|
|
detector = MODELS / "PP-OCRv5_mobile_det"
|
|
recognizer = MODELS / "latin_PP-OCRv5_mobile_rec"
|
|
if not (detector / "inference.pdiparams").is_file() or not (recognizer / "inference.pdiparams").is_file():
|
|
raise SystemExit(3)
|
|
from paddleocr import PaddleOCR
|
|
|
|
engine = PaddleOCR(
|
|
text_detection_model_name="PP-OCRv5_mobile_det",
|
|
text_detection_model_dir=str(detector),
|
|
text_recognition_model_name="latin_PP-OCRv5_mobile_rec",
|
|
text_recognition_model_dir=str(recognizer),
|
|
use_doc_orientation_classify=False,
|
|
use_doc_unwarping=False,
|
|
use_textline_orientation=False,
|
|
text_det_limit_side_len=2048 if args.fine_text else 960,
|
|
text_det_limit_type="max",
|
|
device="cpu",
|
|
)
|
|
results = list(engine.predict(str(image)))
|
|
if len(results) != 1:
|
|
raise RuntimeError("expected_one_page")
|
|
raw = results[0].json["res"]
|
|
lines = []
|
|
for text, polygon in zip(raw["rec_texts"], raw["rec_polys"]):
|
|
points = [[int(x), int(y)] for x, y in polygon]
|
|
xs, ys = zip(*points)
|
|
lines.append({"text": str(text), "bbox": [min(xs), min(ys), max(xs), max(ys)]})
|
|
output.write_text(json.dumps({"lines": lines}, ensure_ascii=False))
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|