mirror of
https://github.com/DS4SD/docling.git
synced 2025-07-27 04:24:45 +00:00
fix: Rename the tesseract OCR related classes and filenames
Signed-off-by: Nikos Livathinos <nli@zurich.ibm.com>
This commit is contained in:
parent
70a8a2cc82
commit
bb8cd0f7fc
@ -17,8 +17,8 @@ from docling.datamodel.document import ConversionResult, DocumentConversionInput
|
|||||||
from docling.datamodel.pipeline_options import (
|
from docling.datamodel.pipeline_options import (
|
||||||
EasyOcrOptions,
|
EasyOcrOptions,
|
||||||
PipelineOptions,
|
PipelineOptions,
|
||||||
TesseractCLIOptions,
|
TesseractCliOcrOptions,
|
||||||
TesseractOptions,
|
TesseractOcrOptions,
|
||||||
)
|
)
|
||||||
from docling.document_converter import DocumentConverter
|
from docling.document_converter import DocumentConverter
|
||||||
|
|
||||||
@ -210,9 +210,9 @@ def convert(
|
|||||||
case OcrEngine.EASYOCR:
|
case OcrEngine.EASYOCR:
|
||||||
ocr_options = EasyOcrOptions()
|
ocr_options = EasyOcrOptions()
|
||||||
case OcrEngine.TESSERACT_CLI:
|
case OcrEngine.TESSERACT_CLI:
|
||||||
ocr_options = TesseractCLIOptions()
|
ocr_options = TesseractCliOcrOptions()
|
||||||
case OcrEngine.TESSERACT:
|
case OcrEngine.TESSERACT:
|
||||||
ocr_options = TesseractOptions()
|
ocr_options = TesseractOcrOptions()
|
||||||
case _:
|
case _:
|
||||||
raise RuntimeError(f"Unexpected backend type {backend}")
|
raise RuntimeError(f"Unexpected backend type {backend}")
|
||||||
|
|
||||||
|
@ -36,7 +36,7 @@ class EasyOcrOptions(OcrOptions):
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
class TesseractCLIOptions(OcrOptions):
|
class TesseractCliOcrOptions(OcrOptions):
|
||||||
kind: Literal["tesseract"] = "tesseract"
|
kind: Literal["tesseract"] = "tesseract"
|
||||||
lang: List[str] = ["fra", "deu", "spa", "eng"]
|
lang: List[str] = ["fra", "deu", "spa", "eng"]
|
||||||
tesseract_cmd: str = "tesseract"
|
tesseract_cmd: str = "tesseract"
|
||||||
@ -47,7 +47,7 @@ class TesseractCLIOptions(OcrOptions):
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
class TesseractOptions(OcrOptions):
|
class TesseractOcrOptions(OcrOptions):
|
||||||
kind: Literal["tesserocr"] = "tesserocr"
|
kind: Literal["tesserocr"] = "tesserocr"
|
||||||
lang: List[str] = ["fra", "deu", "spa", "eng"]
|
lang: List[str] = ["fra", "deu", "spa", "eng"]
|
||||||
path: Optional[str] = None
|
path: Optional[str] = None
|
||||||
@ -62,6 +62,6 @@ class PipelineOptions(BaseModel):
|
|||||||
do_ocr: bool = True # True: perform OCR, replace programmatic PDF text
|
do_ocr: bool = True # True: perform OCR, replace programmatic PDF text
|
||||||
|
|
||||||
table_structure_options: TableStructureOptions = TableStructureOptions()
|
table_structure_options: TableStructureOptions = TableStructureOptions()
|
||||||
ocr_options: Union[EasyOcrOptions, TesseractCLIOptions, TesseractOptions] = Field(
|
ocr_options: Union[EasyOcrOptions, TesseractCliOcrOptions, TesseractOcrOptions] = (
|
||||||
EasyOcrOptions(), discriminator="kind"
|
Field(EasyOcrOptions(), discriminator="kind")
|
||||||
)
|
)
|
||||||
|
@ -7,17 +7,17 @@ from typing import Iterable, Tuple
|
|||||||
import pandas as pd
|
import pandas as pd
|
||||||
|
|
||||||
from docling.datamodel.base_models import BoundingBox, CoordOrigin, OcrCell, Page
|
from docling.datamodel.base_models import BoundingBox, CoordOrigin, OcrCell, Page
|
||||||
from docling.datamodel.pipeline_options import TesseractCLIOptions
|
from docling.datamodel.pipeline_options import TesseractCliOcrOptions
|
||||||
from docling.models.base_ocr_model import BaseOcrModel
|
from docling.models.base_ocr_model import BaseOcrModel
|
||||||
|
|
||||||
_log = logging.getLogger(__name__)
|
_log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
class TesseractCLIModel(BaseOcrModel):
|
class TesseractOcrCliModel(BaseOcrModel):
|
||||||
|
|
||||||
def __init__(self, enabled: bool, options: TesseractCLIOptions):
|
def __init__(self, enabled: bool, options: TesseractCliOcrOptions):
|
||||||
super().__init__(enabled=enabled, options=options)
|
super().__init__(enabled=enabled, options=options)
|
||||||
self.options: TesseractCLIOptions
|
self.options: TesseractCliOcrOptions
|
||||||
|
|
||||||
self.scale = 3 # multiplier for 72 dpi == 216 dpi.
|
self.scale = 3 # multiplier for 72 dpi == 216 dpi.
|
||||||
|
|
@ -4,16 +4,16 @@ from typing import Iterable
|
|||||||
import numpy
|
import numpy
|
||||||
|
|
||||||
from docling.datamodel.base_models import BoundingBox, CoordOrigin, OcrCell, Page
|
from docling.datamodel.base_models import BoundingBox, CoordOrigin, OcrCell, Page
|
||||||
from docling.datamodel.pipeline_options import TesseractCLIOptions
|
from docling.datamodel.pipeline_options import TesseractCliOcrOptions
|
||||||
from docling.models.base_ocr_model import BaseOcrModel
|
from docling.models.base_ocr_model import BaseOcrModel
|
||||||
|
|
||||||
_log = logging.getLogger(__name__)
|
_log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
class TesseractModel(BaseOcrModel):
|
class TesseractOcrModel(BaseOcrModel):
|
||||||
def __init__(self, enabled: bool, options: TesseractCLIOptions):
|
def __init__(self, enabled: bool, options: TesseractCliOcrOptions):
|
||||||
super().__init__(enabled=enabled, options=options)
|
super().__init__(enabled=enabled, options=options)
|
||||||
self.options: TesseractCLIOptions
|
self.options: TesseractCliOcrOptions
|
||||||
|
|
||||||
self.scale = 3 # multiplier for 72 dpi == 216 dpi.
|
self.scale = 3 # multiplier for 72 dpi == 216 dpi.
|
||||||
self.reader = None
|
self.reader = None
|
@ -3,15 +3,15 @@ from pathlib import Path
|
|||||||
from docling.datamodel.pipeline_options import (
|
from docling.datamodel.pipeline_options import (
|
||||||
EasyOcrOptions,
|
EasyOcrOptions,
|
||||||
PipelineOptions,
|
PipelineOptions,
|
||||||
TesseractCLIOptions,
|
TesseractCliOcrOptions,
|
||||||
TesseractOptions,
|
TesseractOcrOptions,
|
||||||
)
|
)
|
||||||
from docling.models.base_ocr_model import BaseOcrModel
|
from docling.models.base_ocr_model import BaseOcrModel
|
||||||
from docling.models.easyocr_model import EasyOcrModel
|
from docling.models.easyocr_model import EasyOcrModel
|
||||||
from docling.models.layout_model import LayoutModel
|
from docling.models.layout_model import LayoutModel
|
||||||
from docling.models.table_structure_model import TableStructureModel
|
from docling.models.table_structure_model import TableStructureModel
|
||||||
from docling.models.tesseract_cli_model import TesseractCLIModel
|
from docling.models.tesseract_ocr_cli_model import TesseractOcrCliModel
|
||||||
from docling.models.tesseract_model import TesseractModel
|
from docling.models.tesseract_ocr_model import TesseractOcrModel
|
||||||
from docling.pipeline.base_model_pipeline import BaseModelPipeline
|
from docling.pipeline.base_model_pipeline import BaseModelPipeline
|
||||||
|
|
||||||
|
|
||||||
@ -28,13 +28,13 @@ class StandardModelPipeline(BaseModelPipeline):
|
|||||||
enabled=pipeline_options.do_ocr,
|
enabled=pipeline_options.do_ocr,
|
||||||
options=pipeline_options.ocr_options,
|
options=pipeline_options.ocr_options,
|
||||||
)
|
)
|
||||||
elif isinstance(pipeline_options.ocr_options, TesseractCLIOptions):
|
elif isinstance(pipeline_options.ocr_options, TesseractCliOcrOptions):
|
||||||
ocr_model = TesseractCLIModel(
|
ocr_model = TesseractOcrCliModel(
|
||||||
enabled=pipeline_options.do_ocr,
|
enabled=pipeline_options.do_ocr,
|
||||||
options=pipeline_options.ocr_options,
|
options=pipeline_options.ocr_options,
|
||||||
)
|
)
|
||||||
elif isinstance(pipeline_options.ocr_options, TesseractOptions):
|
elif isinstance(pipeline_options.ocr_options, TesseractOcrOptions):
|
||||||
ocr_model = TesseractModel(
|
ocr_model = TesseractOcrModel(
|
||||||
enabled=pipeline_options.do_ocr,
|
enabled=pipeline_options.do_ocr,
|
||||||
options=pipeline_options.ocr_options,
|
options=pipeline_options.ocr_options,
|
||||||
)
|
)
|
||||||
|
@ -8,7 +8,10 @@ from docling.backend.docling_parse_backend import DoclingParseDocumentBackend
|
|||||||
from docling.backend.pypdfium2_backend import PyPdfiumDocumentBackend
|
from docling.backend.pypdfium2_backend import PyPdfiumDocumentBackend
|
||||||
from docling.datamodel.base_models import ConversionStatus, PipelineOptions
|
from docling.datamodel.base_models import ConversionStatus, PipelineOptions
|
||||||
from docling.datamodel.document import ConversionResult, DocumentConversionInput
|
from docling.datamodel.document import ConversionResult, DocumentConversionInput
|
||||||
from docling.datamodel.pipeline_options import TesseractCLIOptions, TesseractOptions
|
from docling.datamodel.pipeline_options import (
|
||||||
|
TesseractCliOcrOptions,
|
||||||
|
TesseractOcrOptions,
|
||||||
|
)
|
||||||
from docling.document_converter import DocumentConverter
|
from docling.document_converter import DocumentConverter
|
||||||
|
|
||||||
_log = logging.getLogger(__name__)
|
_log = logging.getLogger(__name__)
|
||||||
@ -126,7 +129,7 @@ def main():
|
|||||||
pipeline_options.do_ocr = True
|
pipeline_options.do_ocr = True
|
||||||
pipeline_options.do_table_structure = True
|
pipeline_options.do_table_structure = True
|
||||||
pipeline_options.table_structure_options.do_cell_matching = True
|
pipeline_options.table_structure_options.do_cell_matching = True
|
||||||
pipeline_options.ocr_options = TesseractOptions()
|
pipeline_options.ocr_options = TesseractOcrOptions()
|
||||||
|
|
||||||
# Docling Parse with Tesseract CLI
|
# Docling Parse with Tesseract CLI
|
||||||
# ----------------------
|
# ----------------------
|
||||||
@ -134,7 +137,7 @@ def main():
|
|||||||
pipeline_options.do_ocr = True
|
pipeline_options.do_ocr = True
|
||||||
pipeline_options.do_table_structure = True
|
pipeline_options.do_table_structure = True
|
||||||
pipeline_options.table_structure_options.do_cell_matching = True
|
pipeline_options.table_structure_options.do_cell_matching = True
|
||||||
pipeline_options.ocr_options = TesseractCLIOptions()
|
pipeline_options.ocr_options = TesseractCliOcrOptions()
|
||||||
|
|
||||||
doc_converter = DocumentConverter(
|
doc_converter = DocumentConverter(
|
||||||
pipeline_options=pipeline_options,
|
pipeline_options=pipeline_options,
|
||||||
|
@ -7,8 +7,8 @@ from docling.datamodel.pipeline_options import (
|
|||||||
EasyOcrOptions,
|
EasyOcrOptions,
|
||||||
OcrOptions,
|
OcrOptions,
|
||||||
PipelineOptions,
|
PipelineOptions,
|
||||||
TesseractCLIOptions,
|
TesseractCliOcrOptions,
|
||||||
TesseractOptions,
|
TesseractOcrOptions,
|
||||||
)
|
)
|
||||||
from docling.document_converter import DocumentConverter
|
from docling.document_converter import DocumentConverter
|
||||||
|
|
||||||
@ -74,8 +74,8 @@ def test_e2e_conversions():
|
|||||||
|
|
||||||
engines: List[OcrOptions] = [
|
engines: List[OcrOptions] = [
|
||||||
EasyOcrOptions(),
|
EasyOcrOptions(),
|
||||||
TesseractOptions(),
|
TesseractOcrOptions(),
|
||||||
TesseractCLIOptions(),
|
TesseractCliOcrOptions(),
|
||||||
]
|
]
|
||||||
|
|
||||||
for ocr_options in engines:
|
for ocr_options in engines:
|
||||||
|
Loading…
Reference in New Issue
Block a user