{"library":"ocrmypdf","type":"library","category":null,"description":"OCRmyPDF is a Python library and application that adds an invisible OCR text layer to scanned PDF files, making them searchable. It utilizes the Tesseract OCR engine and other external tools to process documents, capable of producing highly optimized and archived-ready (PDF/A) files. The project is actively maintained with frequent updates, typically seeing major version releases annually and minor/patch releases more often.","language":"python","status":"active","version":"17.4.1","tags":["pdf","ocr","document-processing","automation","tesseract","pdf/a"],"install":[{"cmd":"pip install ocrmypdf","imports":["from ocrmypdf import ocr","from ocrmypdf import OcrOptions"]}],"homepage":"https://ocrmypdf.readthedocs.io","github":"https://github.com/ocrmypdf/OCRmyPDF","docs":"https://ocrmypdf.readthedocs.io/","changelog":"https://github.com/ocrmypdf/OCRmyPDF/tree/main/docs/releasenotes","pypi":"https://pypi.org/project/ocrmypdf/","npm":null,"openapi_spec":null,"status_page":null,"smithery":null,"compatibility":{"summary":{"python_range":"3.10–3.9","success_rate":100,"avg_install_s":8.5,"avg_import_s":1.75,"wheel_type":"wheel"},"url":"https://checklist.day/v1/registry/ocrmypdf/compatibility"},"provenance":{"verified_status":"passing","verified_at":"Sun Jun 28","last_verified":"Sun Jun 28","next_check":"Tue Jul 28","install_tag":null}}