mirror of
https://github.com/docling-project/docling-eval.git
synced 2026-05-17 13:10:47 +00:00
* fix: Make CVAT pipeline resilient to single document crashes, report failures at the end Signed-off-by: Christoph Auer <cau@zurich.ibm.com> * refactor(cvat_tools): add DocumentStructure accessors and migrate callers - Add O(1) element/path indices and accessor/query APIs on DocumentStructure - Update cvat_to_docling and validator to stop accessing path_mappings/tree_roots/path_to_container directly - Fix PDF -> CVAT <image name> default (doc_{pdf_stem}_page_000001.png) in convert_cvat_to_docling - Update CVAT tests to use accessors and adjust converter test to use CVAT image names Signed-off-by: Christoph Auer <cau@zurich.ibm.com> * feat(cvat): support rotated CVAT boxes via enclosing axis-aligned bbox - Store CVAT rotation on elements (rotation_deg, bbox_unrotated) - Compute enclosing bbox for rotated rectangles and apply during parsing - Add tests for rotation math and parser integration Signed-off-by: Christoph Auer <cau@zurich.ibm.com> * fix: Correct DPI scale in page image Signed-off-by: Christoph Auer <cau@zurich.ibm.com> --------- Signed-off-by: Christoph Auer <cau@zurich.ibm.com>
88 lines
2.7 KiB
Python
88 lines
2.7 KiB
Python
from __future__ import annotations
|
|
|
|
import logging
|
|
from io import BytesIO
|
|
from pathlib import Path
|
|
from typing import Iterator, List
|
|
|
|
from docling_core.types import DoclingDocument
|
|
from docling_core.types.io import DocumentStream
|
|
from pydantic import ValidationError
|
|
|
|
from docling_eval.datamodels.dataset_record import DatasetRecord
|
|
from docling_eval.datamodels.types import BenchMarkColumns
|
|
from docling_eval.utils.utils import extract_images, get_binhash
|
|
|
|
_LOGGER = logging.getLogger(__name__)
|
|
|
|
|
|
def _select_range(files: List[Path], begin_index: int, end_index: int) -> List[Path]:
|
|
if begin_index < 0:
|
|
begin_index = 0
|
|
|
|
total = len(files)
|
|
effective_end = total if end_index < 0 or end_index > total else end_index
|
|
|
|
if begin_index >= effective_end:
|
|
return []
|
|
|
|
return files[begin_index:effective_end]
|
|
|
|
|
|
def iter_docling_json_records(
|
|
json_dir: Path,
|
|
*,
|
|
begin_index: int = 0,
|
|
end_index: int = -1,
|
|
) -> Iterator[DatasetRecord]:
|
|
json_files: List[Path] = sorted(json_dir.glob("*.json"))
|
|
selected_files = _select_range(json_files, begin_index, end_index)
|
|
|
|
for json_path in selected_files:
|
|
try:
|
|
document: DoclingDocument = DoclingDocument.load_from_json(json_path)
|
|
except ValidationError as exc:
|
|
_LOGGER.error(
|
|
"Validation error loading document %s: %s. Skipping this document.",
|
|
json_path,
|
|
exc,
|
|
)
|
|
continue
|
|
except Exception as exc: # noqa: BLE001
|
|
_LOGGER.error(
|
|
"Unexpected error loading document %s: %s. Skipping this document.",
|
|
json_path,
|
|
exc,
|
|
)
|
|
continue
|
|
|
|
try:
|
|
document, pictures, page_images = extract_images(
|
|
document=document,
|
|
pictures_column=BenchMarkColumns.GROUNDTRUTH_PICTURES.value,
|
|
page_images_column=BenchMarkColumns.GROUNDTRUTH_PAGE_IMAGES.value,
|
|
)
|
|
|
|
doc_bytes = json_path.read_bytes()
|
|
|
|
yield DatasetRecord(
|
|
doc_id=json_path.stem,
|
|
doc_path=json_path,
|
|
doc_hash=get_binhash(doc_bytes),
|
|
ground_truth_doc=document,
|
|
ground_truth_pictures=pictures,
|
|
ground_truth_page_images=page_images,
|
|
original=DocumentStream(
|
|
name=json_path.name,
|
|
stream=BytesIO(doc_bytes),
|
|
),
|
|
mime_type="application/json",
|
|
)
|
|
except Exception as exc: # noqa: BLE001
|
|
_LOGGER.error(
|
|
"Error processing document %s: %s. Skipping this document.",
|
|
json_path,
|
|
exc,
|
|
)
|
|
continue
|