fix: Rename the tesseract OCR related classes and filenames

Signed-off-by: Nikos Livathinos <nli@zurich.ibm.com>
This commit is contained in:
Nikos Livathinos 2024-10-08 16:46:25 +02:00
parent 70a8a2cc82
commit bb8cd0f7fc
7 changed files with 34 additions and 31 deletions

View File

@ -17,8 +17,8 @@ from docling.datamodel.document import ConversionResult, DocumentConversionInput
from docling.datamodel.pipeline_options import ( from docling.datamodel.pipeline_options import (
EasyOcrOptions, EasyOcrOptions,
PipelineOptions, PipelineOptions,
TesseractCLIOptions, TesseractCliOcrOptions,
TesseractOptions, TesseractOcrOptions,
) )
from docling.document_converter import DocumentConverter from docling.document_converter import DocumentConverter
@ -210,9 +210,9 @@ def convert(
case OcrEngine.EASYOCR: case OcrEngine.EASYOCR:
ocr_options = EasyOcrOptions() ocr_options = EasyOcrOptions()
case OcrEngine.TESSERACT_CLI: case OcrEngine.TESSERACT_CLI:
ocr_options = TesseractCLIOptions() ocr_options = TesseractCliOcrOptions()
case OcrEngine.TESSERACT: case OcrEngine.TESSERACT:
ocr_options = TesseractOptions() ocr_options = TesseractOcrOptions()
case _: case _:
raise RuntimeError(f"Unexpected backend type {backend}") raise RuntimeError(f"Unexpected backend type {backend}")

View File

@ -36,7 +36,7 @@ class EasyOcrOptions(OcrOptions):
) )
class TesseractCLIOptions(OcrOptions): class TesseractCliOcrOptions(OcrOptions):
kind: Literal["tesseract"] = "tesseract" kind: Literal["tesseract"] = "tesseract"
lang: List[str] = ["fra", "deu", "spa", "eng"] lang: List[str] = ["fra", "deu", "spa", "eng"]
tesseract_cmd: str = "tesseract" tesseract_cmd: str = "tesseract"
@ -47,7 +47,7 @@ class TesseractCLIOptions(OcrOptions):
) )
class TesseractOptions(OcrOptions): class TesseractOcrOptions(OcrOptions):
kind: Literal["tesserocr"] = "tesserocr" kind: Literal["tesserocr"] = "tesserocr"
lang: List[str] = ["fra", "deu", "spa", "eng"] lang: List[str] = ["fra", "deu", "spa", "eng"]
path: Optional[str] = None path: Optional[str] = None
@ -62,6 +62,6 @@ class PipelineOptions(BaseModel):
do_ocr: bool = True # True: perform OCR, replace programmatic PDF text do_ocr: bool = True # True: perform OCR, replace programmatic PDF text
table_structure_options: TableStructureOptions = TableStructureOptions() table_structure_options: TableStructureOptions = TableStructureOptions()
ocr_options: Union[EasyOcrOptions, TesseractCLIOptions, TesseractOptions] = Field( ocr_options: Union[EasyOcrOptions, TesseractCliOcrOptions, TesseractOcrOptions] = (
EasyOcrOptions(), discriminator="kind" Field(EasyOcrOptions(), discriminator="kind")
) )

View File

@ -7,17 +7,17 @@ from typing import Iterable, Tuple
import pandas as pd import pandas as pd
from docling.datamodel.base_models import BoundingBox, CoordOrigin, OcrCell, Page from docling.datamodel.base_models import BoundingBox, CoordOrigin, OcrCell, Page
from docling.datamodel.pipeline_options import TesseractCLIOptions from docling.datamodel.pipeline_options import TesseractCliOcrOptions
from docling.models.base_ocr_model import BaseOcrModel from docling.models.base_ocr_model import BaseOcrModel
_log = logging.getLogger(__name__) _log = logging.getLogger(__name__)
class TesseractCLIModel(BaseOcrModel): class TesseractOcrCliModel(BaseOcrModel):
def __init__(self, enabled: bool, options: TesseractCLIOptions): def __init__(self, enabled: bool, options: TesseractCliOcrOptions):
super().__init__(enabled=enabled, options=options) super().__init__(enabled=enabled, options=options)
self.options: TesseractCLIOptions self.options: TesseractCliOcrOptions
self.scale = 3 # multiplier for 72 dpi == 216 dpi. self.scale = 3 # multiplier for 72 dpi == 216 dpi.

View File

@ -4,16 +4,16 @@ from typing import Iterable
import numpy import numpy
from docling.datamodel.base_models import BoundingBox, CoordOrigin, OcrCell, Page from docling.datamodel.base_models import BoundingBox, CoordOrigin, OcrCell, Page
from docling.datamodel.pipeline_options import TesseractCLIOptions from docling.datamodel.pipeline_options import TesseractCliOcrOptions
from docling.models.base_ocr_model import BaseOcrModel from docling.models.base_ocr_model import BaseOcrModel
_log = logging.getLogger(__name__) _log = logging.getLogger(__name__)
class TesseractModel(BaseOcrModel): class TesseractOcrModel(BaseOcrModel):
def __init__(self, enabled: bool, options: TesseractCLIOptions): def __init__(self, enabled: bool, options: TesseractCliOcrOptions):
super().__init__(enabled=enabled, options=options) super().__init__(enabled=enabled, options=options)
self.options: TesseractCLIOptions self.options: TesseractCliOcrOptions
self.scale = 3 # multiplier for 72 dpi == 216 dpi. self.scale = 3 # multiplier for 72 dpi == 216 dpi.
self.reader = None self.reader = None

View File

@ -3,15 +3,15 @@ from pathlib import Path
from docling.datamodel.pipeline_options import ( from docling.datamodel.pipeline_options import (
EasyOcrOptions, EasyOcrOptions,
PipelineOptions, PipelineOptions,
TesseractCLIOptions, TesseractCliOcrOptions,
TesseractOptions, TesseractOcrOptions,
) )
from docling.models.base_ocr_model import BaseOcrModel from docling.models.base_ocr_model import BaseOcrModel
from docling.models.easyocr_model import EasyOcrModel from docling.models.easyocr_model import EasyOcrModel
from docling.models.layout_model import LayoutModel from docling.models.layout_model import LayoutModel
from docling.models.table_structure_model import TableStructureModel from docling.models.table_structure_model import TableStructureModel
from docling.models.tesseract_cli_model import TesseractCLIModel from docling.models.tesseract_ocr_cli_model import TesseractOcrCliModel
from docling.models.tesseract_model import TesseractModel from docling.models.tesseract_ocr_model import TesseractOcrModel
from docling.pipeline.base_model_pipeline import BaseModelPipeline from docling.pipeline.base_model_pipeline import BaseModelPipeline
@ -28,13 +28,13 @@ class StandardModelPipeline(BaseModelPipeline):
enabled=pipeline_options.do_ocr, enabled=pipeline_options.do_ocr,
options=pipeline_options.ocr_options, options=pipeline_options.ocr_options,
) )
elif isinstance(pipeline_options.ocr_options, TesseractCLIOptions): elif isinstance(pipeline_options.ocr_options, TesseractCliOcrOptions):
ocr_model = TesseractCLIModel( ocr_model = TesseractOcrCliModel(
enabled=pipeline_options.do_ocr, enabled=pipeline_options.do_ocr,
options=pipeline_options.ocr_options, options=pipeline_options.ocr_options,
) )
elif isinstance(pipeline_options.ocr_options, TesseractOptions): elif isinstance(pipeline_options.ocr_options, TesseractOcrOptions):
ocr_model = TesseractModel( ocr_model = TesseractOcrModel(
enabled=pipeline_options.do_ocr, enabled=pipeline_options.do_ocr,
options=pipeline_options.ocr_options, options=pipeline_options.ocr_options,
) )

View File

@ -8,7 +8,10 @@ from docling.backend.docling_parse_backend import DoclingParseDocumentBackend
from docling.backend.pypdfium2_backend import PyPdfiumDocumentBackend from docling.backend.pypdfium2_backend import PyPdfiumDocumentBackend
from docling.datamodel.base_models import ConversionStatus, PipelineOptions from docling.datamodel.base_models import ConversionStatus, PipelineOptions
from docling.datamodel.document import ConversionResult, DocumentConversionInput from docling.datamodel.document import ConversionResult, DocumentConversionInput
from docling.datamodel.pipeline_options import TesseractCLIOptions, TesseractOptions from docling.datamodel.pipeline_options import (
TesseractCliOcrOptions,
TesseractOcrOptions,
)
from docling.document_converter import DocumentConverter from docling.document_converter import DocumentConverter
_log = logging.getLogger(__name__) _log = logging.getLogger(__name__)
@ -126,7 +129,7 @@ def main():
pipeline_options.do_ocr = True pipeline_options.do_ocr = True
pipeline_options.do_table_structure = True pipeline_options.do_table_structure = True
pipeline_options.table_structure_options.do_cell_matching = True pipeline_options.table_structure_options.do_cell_matching = True
pipeline_options.ocr_options = TesseractOptions() pipeline_options.ocr_options = TesseractOcrOptions()
# Docling Parse with Tesseract CLI # Docling Parse with Tesseract CLI
# ---------------------- # ----------------------
@ -134,7 +137,7 @@ def main():
pipeline_options.do_ocr = True pipeline_options.do_ocr = True
pipeline_options.do_table_structure = True pipeline_options.do_table_structure = True
pipeline_options.table_structure_options.do_cell_matching = True pipeline_options.table_structure_options.do_cell_matching = True
pipeline_options.ocr_options = TesseractCLIOptions() pipeline_options.ocr_options = TesseractCliOcrOptions()
doc_converter = DocumentConverter( doc_converter = DocumentConverter(
pipeline_options=pipeline_options, pipeline_options=pipeline_options,

View File

@ -7,8 +7,8 @@ from docling.datamodel.pipeline_options import (
EasyOcrOptions, EasyOcrOptions,
OcrOptions, OcrOptions,
PipelineOptions, PipelineOptions,
TesseractCLIOptions, TesseractCliOcrOptions,
TesseractOptions, TesseractOcrOptions,
) )
from docling.document_converter import DocumentConverter from docling.document_converter import DocumentConverter
@ -74,8 +74,8 @@ def test_e2e_conversions():
engines: List[OcrOptions] = [ engines: List[OcrOptions] = [
EasyOcrOptions(), EasyOcrOptions(),
TesseractOptions(), TesseractOcrOptions(),
TesseractCLIOptions(), TesseractCliOcrOptions(),
] ]
for ocr_options in engines: for ocr_options in engines: