Files
docling-eval/docling_eval/utils/external_docling_document_loader.py
Nikos Livathinos 9d04a56b93 feat: Parallelize the evaluation of tables and cache the loading of external predictions (#190)
* feat: Extend evaluate_dpbench_on_external_predictions.sh to include visualisations of the evaluations

Signed-off-by: Nikos Livathinos <nli@zurich.ibm.com>

* fix: Improve error checking in main.py:visualize()

Signed-off-by: Nikos Livathinos <nli@zurich.ibm.com>

* fix: Improve logging

Signed-off-by: Nikos Livathinos <nli@zurich.ibm.com>

* feat: Parallelize the computation of PixelLayoutEvaluator at the level of page

Signed-off-by: Nikos Livathinos <nli@zurich.ibm.com>

* fix: Make DatasetPixelLayoutEvaluation a subclass of DatasetEvaluation

Signed-off-by: Nikos Livathinos <nli@zurich.ibm.com>

* feat: Parallelize the MarkdownTextEvaluator

Signed-off-by: Nikos Livathinos <nli@zurich.ibm.com>

* chore: Improve logging

Signed-off-by: Nikos Livathinos <nli@zurich.ibm.com>

* feat: Parallelize the table evaluation

Signed-off-by: Nikos Livathinos <nli@zurich.ibm.com>

* fix: Enclose evaluate_tables() in try catch

Signed-off-by: Nikos Livathinos <nli@zurich.ibm.com>

* fix: Improve logging in TableEvaluator

Signed-off-by: Nikos Livathinos <nli@zurich.ibm.com>

* fix: Fix loop counter in TableEvaluator and initialize with the correct concurrency

Signed-off-by: Nikos Livathinos <nli@zurich.ibm.com>

* feat: Refactor ExternalDoclingDocumentLoader to enable caching of loaded documents.
Refactor main to initialize a single object of the external loader and refactor all evaluators to
receive the loader instead of the raw Path with external predictions.

Signed-off-by: Nikos Livathinos <nli@zurich.ibm.com>

* fix: Improve logging in ExternalDoclingDocumentLoader

Signed-off-by: Nikos Livathinos <nli@zurich.ibm.com>

* feat: Extend the evaluate() and the CLI to support multiple modalities and multiple evaluation results:
- Refactor the codebase to support the new signature of evaluate()
- All modalities share a common ExternalDoclingDocumentLoader object with caching.
- Support multiple evaluation results to keep results for all evaluation metrics

Signed-off-by: Nikos Livathinos <nli@zurich.ibm.com>

* chore: Improve logging in evaluators

Signed-off-by: Nikos Livathinos <nli@zurich.ibm.com>

* fix: Fix the log level for some messages

Signed-off-by: Nikos Livathinos <nli@zurich.ibm.com>

* fix: Fix setting the max_new_tokens parameter when initializing GraniteDocling

Signed-off-by: Nikos Livathinos <nli@zurich.ibm.com>

* fix: Improve API. Code styling

Signed-off-by: Nikos Livathinos <nli@zurich.ibm.com>

---------

Signed-off-by: Nikos Livathinos <nli@zurich.ibm.com>
2025-12-19 12:51:40 +01:00

173 lines
6.0 KiB
Python

import logging
from pathlib import Path
from typing import Optional
from docling_core.types.doc.document import (
DoclingDocument,
DocTagsDocument,
DocTagsPage,
)
from PIL import Image
from docling_eval.datamodels.dataset_record import DatasetRecord
_log = logging.getLogger(__name__)
class ExternalDoclingDocumentLoader:
r""" """
def __init__(
self,
external_predictions_dir: Path,
enable_cache: bool = True,
):
r""" """
self._external_predictions_dir = external_predictions_dir
self._enable_cache = enable_cache
# doc-id -> DoclingDocument
self._cache: dict[str, DoclingDocument] = {}
def predictions_path(self) -> Path:
r""" """
return self._external_predictions_dir
def get(self, record: DatasetRecord) -> Optional[DoclingDocument]:
r"""
Cache enabled loading of DoclingDocuments
Logic:
1. If the cache is enabled and the doc_id is in cache return it.
2. Otherwise try to load the document and return it.
3. If the document cannot be loaded return None
"""
doc_id = record.doc_id
doc: Optional[DoclingDocument] = None
if self._enable_cache and doc_id in self._cache:
doc = self._cache[doc_id]
else:
doc = self.load(record)
return doc
def load(
self,
record: DatasetRecord,
) -> Optional[DoclingDocument]:
r"""
Load the DoclingDocument from the external predictions path
The following fields are used from the `record` parameter:
- record.doc_id
- record.ground_truth_page_images[0]
"""
doc_id = record.doc_id
json_fn = self._external_predictions_dir / f"{doc_id}.json"
doctags_fn = ExternalDoclingDocumentLoader.build_doctags_path(
self._external_predictions_dir, doc_id
)
yaml_fn = self._external_predictions_dir / f"{doc_id}.yaml"
yml_fn = self._external_predictions_dir / f"{doc_id}.yml"
doc: Optional[DoclingDocument] = None
if json_fn.is_file():
doc = DoclingDocument.load_from_json(json_fn)
elif doctags_fn.is_file():
gt_page_images = record.ground_truth_page_images
gt_page_image = gt_page_images[0] if len(gt_page_images) > 0 else None
doc = ExternalDoclingDocumentLoader.load_doctags(
doc_id,
self._external_predictions_dir,
gt_page_image=gt_page_image,
)
elif yaml_fn.is_file():
doc = DoclingDocument.load_from_yaml(yaml_fn)
elif yml_fn.is_file():
doc = DoclingDocument.load_from_yaml(yml_fn)
if doc is None:
_log.error("Failed to load document: %s", doc_id)
# Check if to update the cache
if self._enable_cache and doc is not None:
_log.debug("Caching externally loaded document: %s", doc_id)
self._cache[doc_id] = doc
return doc
@staticmethod
def build_doctags_path(doctags_root: Path, doc_id: str) -> Path:
r"""Get the full path of the doctags file"""
dt_path = doctags_root / f"{doc_id}.dt"
return dt_path
@staticmethod
def load_doctags(
doc_id: str,
doctags_root: Path,
page_images_root: Optional[Path] = None,
gt_page_image: Optional[Image.Image] = None,
image_filename_extension: str = "png",
) -> Optional[DoclingDocument]:
r"""
Load a single page DoclingDocument object from a doctags file and a page image.
The page image is supplied from these sources in the specific order:
1. The page_images_root: An image with filename <doc_id>.<image_filename_extension> is used
2. gt_page_image: An explicit Image object is used.
3. Search for the image with filename <doc_id>.<image_filename_extension> in the doctags root
Parameters
----------
doctags_root: Root path to load doctags as files with name <doc_id>.dt
doc_id: The document id of the file to be loaded
page_images_root: If provided, search for the page images here first.
gt_page_image: If provided, search use that object for the page image.
image_filename_extension: The file extension for the page image.
Returns
-------
DoclingDocument object or None if the document cannot be reconstructed
"""
# Read the doctags file
doctags_fn = ExternalDoclingDocumentLoader.build_doctags_path(
doctags_root, doc_id
)
try:
with open(doctags_fn, "r") as fd:
doctags = fd.read()
page_image: Optional[Image.Image] = None
if page_images_root:
page_image_fn = (
page_images_root / f"{doc_id}.{image_filename_extension}"
)
if page_image_fn.is_file():
page_image = Image.open(page_image_fn)
else:
_log.warning("Failed to load page image: %s", page_image_fn)
elif gt_page_image is not None:
page_image = gt_page_image
else:
page_image_fn = doctags_root / f"{doc_id}.{image_filename_extension}"
if page_image_fn.is_file():
page_image = Image.open(page_image_fn)
else:
_log.warning(
"Missing page image file: %s. Reconstruct doctags without page image",
page_image_fn,
)
# Build DoclingDocument
doctags_page = DocTagsPage(tokens=doctags, image=page_image)
doctags_doc = DocTagsDocument(pages=[doctags_page])
doc = DoclingDocument.load_from_doctags(doctags_doc, document_name=doc_id)
return doc
except Exception as e:
_log.error(f"Error loading doctags document {doc_id}: {str(e)}")
return None