Files
docling-core/test/test_collection.py
c73904e68e style: replace black, isort, flake8 and autoflake with ruff (#456)
* Added ruff to dev dependencies

Signed-off-by: Florian Schwarb <florian.schwarb@gmail.com>

* Added ruff settings to pyproject.toml as in docling

Signed-off-by: Florian Schwarb <florian.schwarb@gmail.com>

* Cleanup uf pyproject.toml

Signed-off-by: Florian Schwarb <florian.schwarb@gmail.com>

* Copied settings for ruff pre-commit hooks from docling

Signed-off-by: Florian Schwarb <florian.schwarb@gmail.com>

* Excluded test/data/** from ruff formatting / linting

Signed-off-by: Florian Schwarb <florian.schwarb@gmail.com>

* ruff format

Signed-off-by: Florian Schwarb <florian.schwarb@gmail.com>

* Added some ignore statements to pyproject.toml such that ruff check raises fewer issues

Signed-off-by: Florian Schwarb <florian.schwarb@gmail.com>

* ruff check --fix

Signed-off-by: Florian Schwarb <florian.schwarb@gmail.com>

* Ignored some more rules

Signed-off-by: Florian Schwarb <florian.schwarb@gmail.com>

* Fixed the rest of the errors that would only concern 1 - 3 files

Signed-off-by: Florian Schwarb <florian.schwarb@gmail.com>

* Added another ignore related to df for DataFrame names

Signed-off-by: Florian Schwarb <florian.schwarb@gmail.com>

* Modified CONTRIBUTING.md such that black / isort are replaced by ruff

Signed-off-by: Florian Schwarb <florian.schwarb@gmail.com>

* Added UP045 to ignore list such that Optional[...] does not raise

Signed-off-by: Florian Schwarb <florian.schwarb@gmail.com>

* Moved .flake8 configs to pyproject.toml

Signed-off-by: Florian Schwarb <florian.schwarb@gmail.com>

* Moved autoflake to be used with ruff

Signed-off-by: Florian Schwarb <florian.schwarb@gmail.com>

* Moved all .flake8 settings to pyproject.toml to be compatible with ruff (i.e. no separate [tool.flake8] section

Signed-off-by: Florian Schwarb <florian.schwarb@gmail.com>

* Removed flake8 from .pre-commit hooks

Signed-off-by: Florian Schwarb <florian.schwarb@gmail.com>

* Applied ruff format (again); formatted some files as the line-length = 120 equals now what was set for the .flake8 settings

Signed-off-by: Florian Schwarb <florian.schwarb@gmail.com>

* Set max-complexity to 30 (as was originally) in the pyproject.toml as one linting check would fail

Signed-off-by: Florian Schwarb <florian.schwarb@gmail.com>

* Adding PD901 to ignore list such that pre-commit hooks run fully again

Signed-off-by: Florian Schwarb <florian.schwarb@gmail.com>

* Replaced dtype | None syntax by Optional[dtype] in remaining places

Signed-off-by: Florian Schwarb <florian.schwarb@gmail.com>

* chore: fix 'test' ref in pyproject

Signed-off-by: Cesar Berrospi Ramis <ceb@zurich.ibm.com>

* style: remove typing List, Set, Tuple, Dict

Signed-off-by: Cesar Berrospi Ramis <ceb@zurich.ibm.com>

* style: remove UP015 check from ignore list

Signed-off-by: Cesar Berrospi Ramis <ceb@zurich.ibm.com>

* style: remove UP034 check from ignore list

Signed-off-by: Cesar Berrospi Ramis <ceb@zurich.ibm.com>

* style: normalize dashes in comments and docstrings

Signed-off-by: Cesar Berrospi Ramis <ceb@zurich.ibm.com>

* style: remove PD901 check from ignore list

Signed-off-by: Cesar Berrospi Ramis <ceb@zurich.ibm.com>

* style: remove C403 check from ignore list

Signed-off-by: Cesar Berrospi Ramis <ceb@zurich.ibm.com>

* style: remove C403, C413, C416 check from ignore list

Signed-off-by: Cesar Berrospi Ramis <ceb@zurich.ibm.com>

* style: remove E203, F811 check from ignore list

Signed-off-by: Cesar Berrospi Ramis <ceb@zurich.ibm.com>

---------

Signed-off-by: Florian Schwarb <florian.schwarb@gmail.com>
Signed-off-by: Cesar Berrospi Ramis <ceb@zurich.ibm.com>
Co-authored-by: Florian Schwarb <florian.schwarb@gmail.com>
Co-authored-by: Cesar Berrospi Ramis <ceb@zurich.ibm.com>
2026-01-13 17:03:10 +01:00

143 lines
4.7 KiB
Python

"""Test the pydantic models in module types."""
import glob
import pytest
from pydantic import ValidationError
from docling_core.types import Generic, Record
from docling_core.types.legacy_doc.document import ExportedCCSDocument as Document
GENERATE = False
def test_generic():
"""Test the Generic model."""
input_generic_0 = {
"file-info": {
"filename": "abc.xml",
"filename-prov": "abc.xml.zip",
"document-hash": "123457889",
},
"_name": "The ABC legacy_doc",
"custom": ["The custom ABC content 1.", "The custom ABC content 2."],
}
Generic.model_validate(input_generic_0)
input_generic_1 = {
"file-info": {"filename": "abc.xml", "document-hash": "123457889"},
"_name": "The ABC legacy_doc",
}
Generic.model_validate(input_generic_1)
input_generic_2 = {
"_name": "The ABC legacy_doc",
"custom": ["The custom ABC content 1.", "The custom ABC content 2."],
}
with pytest.raises(ValidationError):
Generic.model_validate(input_generic_2)
def test_document():
"""Test the Document model."""
for filename in glob.glob("test/data/legacy_doc/doc-*.json"):
with open(filename, encoding="utf-8") as file_obj:
file_json = file_obj.read()
Document.model_validate_json(file_json)
def test_table_export_to_tokens():
"""Test the Table Tokens export."""
for filename in glob.glob("test/data/legacy_doc/doc-*.json"):
with open(filename, encoding="utf-8") as file_obj:
file_json = file_obj.read()
doc = Document.model_validate_json(file_json)
if doc.tables is not None and doc.page_dimensions is not None:
pagedims = doc.get_map_to_page_dimensions()
if doc.tables is not None:
for i, table in enumerate(doc.tables):
page = table.prov[0].page
out = table.export_to_document_tokens(page_w=pagedims[page][0], page_h=pagedims[page][1])
fname = f"{filename}_table_{i}.dt.txt"
if GENERATE:
print(f"writing {fname}")
with open(fname, "w", encoding="utf-8") as gold_obj:
gold_obj.write(out)
with open(fname, "r", encoding="utf-8") as gold_obj:
gold_data = gold_obj.read()
assert out == gold_data
# we only test on the first table
break
elif doc.tables is not None and doc.page_dimensions is None:
if doc.tables is not None:
for i, table in enumerate(doc.tables):
page = table.prov[0].page
out = table.export_to_document_tokens(add_table_location=False, add_cell_location=False)
fname = f"{filename}_table_{i}.dt.txt"
if GENERATE:
print(f"writing {fname}")
with open(fname, "w", encoding="utf-8") as gold_obj:
gold_obj.write(out)
with open(fname, "r", encoding="utf-8") as gold_obj:
gold_data = gold_obj.read()
assert out == gold_data
# we only test on the first table
break
def test_document_export_to_md():
"""Test the Document Markdown export."""
with open("test/data/legacy_doc/doc-export.json", encoding="utf-8") as src_obj:
src_data = src_obj.read()
doc = Document.model_validate_json(src_data)
md = doc.export_to_markdown()
if GENERATE:
with open("test/data/legacy_doc/doc-export.md", "w", encoding="utf-8") as gold_obj:
gold_obj.write(md)
with open("test/data/legacy_doc/doc-export.md", encoding="utf-8") as gold_obj:
gold_data = gold_obj.read().strip()
assert md == gold_data
def test_document_export_to_tokens():
"""Test the Document Tokens export."""
with open("test/data/legacy_doc/doc-export.json", encoding="utf-8") as src_obj:
src_data = src_obj.read()
doc = Document.model_validate_json(src_data)
xml = doc.export_to_document_tokens(delim=True)
if GENERATE:
with open("test/data/legacy_doc/doc-export.dt.txt", "w", encoding="utf-8") as gold_obj:
gold_obj.write(xml)
with open("test/data/legacy_doc/doc-export.dt.txt", "r", encoding="utf-8") as gold_obj:
gold_data = gold_obj.read().strip()
assert xml == gold_data
def test_record():
"""Test the Document model."""
for filename in glob.glob("test/data/rec/record-*.json"):
with open(filename, encoding="utf-8") as file_obj:
file_json = file_obj.read()
Record.model_validate_json(file_json)