Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 6 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
@@ -1,3 +1,9 @@
## 0.27.19

### Fixes

- **`partition_tsv()` honours `infer_table_structure`.** The function had no `infer_table_structure` parameter, so the argument was absorbed by `**kwargs` and dropped: `Table.metadata.text_as_html` was set even for `partition(..., infer_table_structure=False)`, or for a `skip_infer_table_types` list containing `tsv`. `partition_csv()`, `partition_xlsx()`, `partition_docx()` and `partition_odt()` all gate that field on the flag, and `decide_table_extraction()` in `partition/auto.py` passes it to every non-special partitioner. The parameter now has the same name, default and docstring as its siblings.

## 0.27.18

### Fixes
Expand Down
24 changes: 24 additions & 0 deletions test_unstructured/partition/test_tsv.py
Original file line number Diff line number Diff line change
Expand Up @@ -23,6 +23,7 @@
from unstructured.common.html_table import HtmlTable
from unstructured.documents.elements import Table
from unstructured.errors import UnprocessableEntityError
from unstructured.partition.auto import partition
from unstructured.partition.tsv import partition_tsv

EXPECTED_FILETYPE = "text/tsv"
Expand Down Expand Up @@ -286,3 +287,26 @@ def test_partition_tsv_with_implicit_index_columns_matches_pandas_within_the_lim
index=False, header=True, na_rep=""
)
assert table.text == HtmlTable.from_html_text(expected).text


# -- `infer_table_structure` ---------------------------------------------------------------------


@pytest.mark.parametrize("infer_table_structure", [True, False])
def test_partition_tsv_from_filename_infer_table_structure(infer_table_structure: bool):
elements = partition_tsv(
example_doc_path("stanley-cups.tsv"), infer_table_structure=infer_table_structure
)

has_text_as_html = elements[0].metadata.text_as_html is not None
assert has_text_as_html == infer_table_structure


@pytest.mark.parametrize("infer_table_structure", [True, False])
def test_partition_tsv_via_partition_respects_infer_table_structure(infer_table_structure: bool):
elements = partition(
example_doc_path("stanley-cups.tsv"), infer_table_structure=infer_table_structure
)

has_text_as_html = elements[0].metadata.text_as_html is not None
assert has_text_as_html == infer_table_structure
2 changes: 1 addition & 1 deletion unstructured/__version__.py
Original file line number Diff line number Diff line change
@@ -1 +1 @@
__version__ = "0.27.18" # pragma: no cover
__version__ = "0.27.19" # pragma: no cover
9 changes: 8 additions & 1 deletion unstructured/partition/tsv.py
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,7 @@ def partition_tsv(
*,
file: Optional[IO[bytes]] = None,
include_header: bool = False,
infer_table_structure: bool = True,
**kwargs: Any,
) -> list[Element]:
"""Partitions TSV files into document elements.
Expand All @@ -42,6 +43,12 @@ def partition_tsv(
A file-like object using "rb" mode --> open(filename, "rb").
include_header
Determines whether or not header info info is included in text and medatada.text_as_html.
infer_table_structure
If True, any Table elements that are extracted will also have a metadata field
named "text_as_html" where the table's text content is rendered into an html string.
I.e., rows and cells are preserved.
Whether True or False, the "text" field is always present in any Table element
and is the text content of the table (no structure).
"""
exactly_one(filename=filename, file=file)

Expand Down Expand Up @@ -72,7 +79,7 @@ def partition_tsv(
metadata = ElementMetadata(
filename=filename,
last_modified=get_last_modified_date(filename) if filename else None,
text_as_html=html_table.html,
text_as_html=html_table.html if infer_table_structure else None,
)
metadata.detection_origin = DETECTION_ORIGIN

Expand Down
Loading