fix: Safe pipeline init, use device_map in transformers models (#1917)

* Use device_map for transformer models Signed-off-by: Christoph Auer <cau@zurich.ibm.com> * Add accelerate Signed-off-by: Christoph Auer <cau@zurich.ibm.com> * Relax accelerate min version Signed-off-by: Christoph Auer <cau@zurich.ibm.com> * Make pipeline cache+init thread-safe Signed-off-by: Christoph Auer <cau@zurich.ibm.com> --------- Signed-off-by: Christoph Auer <cau@zurich.ibm.com>
2025-12-10 13:48:13 +00:00 · 2025-07-18 15:14:36 +02:00
parent e1e3053695
commit cca05c45ea
4 changed files with 19 additions and 12 deletions
--- a/docling/document_converter.py
+++ b/docling/document_converter.py
@@ -1,6 +1,7 @@
 import hashlib
 import logging
 import sys
+import threading
 import time
 from collections.abc import Iterable, Iterator
 from functools import partial
@@ -49,6 +50,7 @@ from docling.pipeline.standard_pdf_pipeline import StandardPdfPipeline
 from docling.utils.utils import chunkify

 _log = logging.getLogger(__name__)
+_PIPELINE_CACHE_LOCK = threading.Lock()


 class FormatOption(BaseModel):
@@ -315,17 +317,18 @@ class DocumentConverter:
        # Use a composite key to cache pipelines
        cache_key = (pipeline_class, options_hash)

-        if cache_key not in self.initialized_pipelines:
-            _log.info(
-                f"Initializing pipeline for {pipeline_class.__name__} with options hash {options_hash}"
-            )
-            self.initialized_pipelines[cache_key] = pipeline_class(
-                pipeline_options=pipeline_options
-            )
-        else:
-            _log.debug(
-                f"Reusing cached pipeline for {pipeline_class.__name__} with options hash {options_hash}"
-            )
+        with _PIPELINE_CACHE_LOCK:
+            if cache_key not in self.initialized_pipelines:
+                _log.info(
+                    f"Initializing pipeline for {pipeline_class.__name__} with options hash {options_hash}"
+                )
+                self.initialized_pipelines[cache_key] = pipeline_class(
+                    pipeline_options=pipeline_options
+                )
+            else:
+                _log.debug(
+                    f"Reusing cached pipeline for {pipeline_class.__name__} with options hash {options_hash}"
+                )

        return self.initialized_pipelines[cache_key]

--- a/docling/models/picture_description_vlm_model.py
+++ b/docling/models/picture_description_vlm_model.py
@@ -65,6 +65,7 @@ class PictureDescriptionVlmModel(
                self.processor = AutoProcessor.from_pretrained(artifacts_path)
                self.model = AutoModelForVision2Seq.from_pretrained(
                    artifacts_path,
+                    device_map=self.device,
                    torch_dtype=torch.bfloat16,
                    _attn_implementation=(
                        "flash_attention_2"
@@ -72,7 +73,7 @@ class PictureDescriptionVlmModel(
                        and accelerator_options.cuda_use_flash_attention2
                        else "eager"
                    ),
-                ).to(self.device)
+                )

            self.provenance = f"{self.options.repo_id}"