mirror of
https://github.com/DS4SD/docling.git
synced 2025-12-08 12:48:28 +00:00
refactor(HTML): handle text from styled html (#1960)
* A new HTML backend that handles styled html (ignors it) as well as images. Images are parsed as placeholders with a caption, if it exists. Co-authored-by: Cesar Berrospi Ramis <75900930+ceberam@users.noreply.github.com> Co-authored-by: vaaale <2428222+vaaale@users.noreply.github.com> Signed-off-by: Alexander Vaagan <alexander.vaagan@gmail.com> Signed-off-by: Cesar Berrospi Ramis <75900930+ceberam@users.noreply.github.com> Signed-off-by: vaaale <2428222+vaaale@users.noreply.github.com> * tests(HTML): re-enable test_ordered_lists Re-enable test_ordered_lists regression test for the HTML backend since docling-core now supports ordered lists with custom start value. Signed-off-by: Cesar Berrospi Ramis <75900930+ceberam@users.noreply.github.com> --------- Signed-off-by: Alexander Vaagan <alexander.vaagan@gmail.com> Signed-off-by: Cesar Berrospi Ramis <75900930+ceberam@users.noreply.github.com> Signed-off-by: vaaale <2428222+vaaale@users.noreply.github.com> Co-authored-by: Alexander Vaagan <2428222+vaaale@users.noreply.github.com>
This commit is contained in:
committed by
GitHub
parent
5d98bcea1b
commit
a069b1175b
@@ -1,8 +1,6 @@
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from docling.backend.html_backend import HTMLDocumentBackend
|
||||
from docling.datamodel.base_models import InputFormat
|
||||
from docling.datamodel.document import (
|
||||
@@ -37,17 +35,15 @@ def test_heading_levels():
|
||||
if isinstance(item, SectionHeaderItem):
|
||||
if item.text == "Etymology":
|
||||
found_lvl_1 = True
|
||||
# h2 becomes level 1 because of h1 as title
|
||||
assert item.level == 1
|
||||
elif item.text == "Feeding":
|
||||
found_lvl_2 = True
|
||||
# h3 becomes level 2 because of h1 as title
|
||||
assert item.level == 2
|
||||
assert found_lvl_1 and found_lvl_2
|
||||
|
||||
|
||||
@pytest.mark.skip(
|
||||
"Temporarily disabled since docling-core>=2.21.0 does not support ordered lists "
|
||||
"with custom start value"
|
||||
)
|
||||
def test_ordered_lists():
|
||||
test_set: list[tuple[bytes, str]] = []
|
||||
|
||||
|
||||
Reference in New Issue
Block a user