fix: parse HTML files without body tag

Parse HTML files without 'body' tag, since it is optional in HTML5 specification. Signed-off-by: Cesar Berrospi Ramis <75900930+ceberam@users.noreply.github.com>
2025-08-02 07:22:14 +00:00 · 2025-01-27 15:12:18 +01:00 · 2025-01-27 15:12:18 +01:00 · baf622ffad
commit baf622ffad
parent 53327552e8
1 changed files with 3 additions and 2 deletions
--- a/docling/backend/html_backend.py
+++ b/docling/backend/html_backend.py
@ -78,10 +78,11 @@ class HTMLDocumentBackend(DeclarativeDocumentBackend):

        if self.is_valid():
            assert self.soup is not None
+            content = self.soup.body or self.soup
            # Replace <br> tags with newline characters
-            for br in self.soup.body.find_all("br"):
+            for br in content.find_all("br"):
                br.replace_with("\n")
-            doc = self.walk(self.soup.body, doc)
+            doc = self.walk(content, doc)
        else:
            raise RuntimeError(
                f"Cannot convert doc with {self.document_hash} because the backend failed to init."