mirror of
https://github.com/DS4SD/docling.git
synced 2025-07-26 20:14:47 +00:00
not do amateur hour stuff
Signed-off-by: Vinay Damodaran <vrdn@hey.com>
This commit is contained in:
parent
45c5a9445a
commit
8186bdcd4c
@ -225,6 +225,7 @@ class PictureDescriptionApiOptions(PictureDescriptionBaseOptions):
|
|||||||
headers: Dict[str, str] = {}
|
headers: Dict[str, str] = {}
|
||||||
params: Dict[str, Any] = {}
|
params: Dict[str, Any] = {}
|
||||||
timeout: float = 20
|
timeout: float = 20
|
||||||
|
concurrency: int = 1
|
||||||
|
|
||||||
prompt: str = "Describe this image in a few sentences."
|
prompt: str = "Describe this image in a few sentences."
|
||||||
provenance: str = ""
|
provenance: str = ""
|
||||||
@ -295,6 +296,7 @@ class ApiVlmOptions(BaseVlmOptions):
|
|||||||
params: Dict[str, Any] = {}
|
params: Dict[str, Any] = {}
|
||||||
scale: float = 2.0
|
scale: float = 2.0
|
||||||
timeout: float = 60
|
timeout: float = 60
|
||||||
|
concurrency: int = 1
|
||||||
response_format: ResponseFormat
|
response_format: ResponseFormat
|
||||||
|
|
||||||
|
|
||||||
|
@ -28,6 +28,7 @@ class ApiVlmModel(BasePageModel):
|
|||||||
)
|
)
|
||||||
|
|
||||||
self.timeout = self.vlm_options.timeout
|
self.timeout = self.vlm_options.timeout
|
||||||
|
self.concurrency = self.vlm_options.concurrency
|
||||||
self.prompt_content = (
|
self.prompt_content = (
|
||||||
f"This is a page from a document.\n{self.vlm_options.prompt}"
|
f"This is a page from a document.\n{self.vlm_options.prompt}"
|
||||||
)
|
)
|
||||||
@ -37,10 +38,7 @@ class ApiVlmModel(BasePageModel):
|
|||||||
}
|
}
|
||||||
|
|
||||||
def __call__(
|
def __call__(
|
||||||
self,
|
self, conv_res: ConversionResult, page_batch: Iterable[Page]
|
||||||
conv_res: ConversionResult,
|
|
||||||
page_batch: Iterable[Page],
|
|
||||||
concurrency: int = 1,
|
|
||||||
) -> Iterable[Page]:
|
) -> Iterable[Page]:
|
||||||
def _vlm_request(page):
|
def _vlm_request(page):
|
||||||
assert page._backend is not None
|
assert page._backend is not None
|
||||||
@ -69,5 +67,5 @@ class ApiVlmModel(BasePageModel):
|
|||||||
|
|
||||||
return page
|
return page
|
||||||
|
|
||||||
with ThreadPoolExecutor(max_workers=concurrency) as executor:
|
with ThreadPoolExecutor(max_workers=self.concurrency) as executor:
|
||||||
yield from executor.map(_vlm_request, page_batch)
|
yield from executor.map(_vlm_request, page_batch)
|
||||||
|
@ -38,6 +38,7 @@ class PictureDescriptionApiModel(PictureDescriptionBaseModel):
|
|||||||
accelerator_options=accelerator_options,
|
accelerator_options=accelerator_options,
|
||||||
)
|
)
|
||||||
self.options: PictureDescriptionApiOptions
|
self.options: PictureDescriptionApiOptions
|
||||||
|
self.concurrency = self.options.concurrency
|
||||||
|
|
||||||
if self.enabled:
|
if self.enabled:
|
||||||
if not enable_remote_services:
|
if not enable_remote_services:
|
||||||
@ -46,9 +47,7 @@ class PictureDescriptionApiModel(PictureDescriptionBaseModel):
|
|||||||
"pipeline_options.enable_remote_services=True."
|
"pipeline_options.enable_remote_services=True."
|
||||||
)
|
)
|
||||||
|
|
||||||
def _annotate_images(
|
def _annotate_images(self, images: Iterable[Image.Image]) -> Iterable[str]:
|
||||||
self, images: Iterable[Image.Image], concurrency: int = 1
|
|
||||||
) -> Iterable[str]:
|
|
||||||
# Note: technically we could make a batch request here,
|
# Note: technically we could make a batch request here,
|
||||||
# but not all APIs will allow for it. For example, vllm won't allow more than 1.
|
# but not all APIs will allow for it. For example, vllm won't allow more than 1.
|
||||||
def _api_request(image):
|
def _api_request(image):
|
||||||
@ -61,5 +60,5 @@ class PictureDescriptionApiModel(PictureDescriptionBaseModel):
|
|||||||
**self.options.params,
|
**self.options.params,
|
||||||
)
|
)
|
||||||
|
|
||||||
with ThreadPoolExecutor(max_workers=concurrency) as executor:
|
with ThreadPoolExecutor(max_workers=self.concurrency) as executor:
|
||||||
yield from executor.map(_api_request, images)
|
yield from executor.map(_api_request, images)
|
||||||
|
Loading…
Reference in New Issue
Block a user