Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 6 additions & 1 deletion CHANGELOG.md
Original file line number Diff line number Diff line change
@@ -1,3 +1,9 @@
## 0.27.18

### Fixes

- **Token-based `chunk_by_title()` fills the chunking window again.** `PreChunk.can_combine()` sized its text with `len()`, so the `combine_text_under_n_chars` threshold and the hard-max check counted characters even when `max_tokens` selected token counting; every section looked too large to combine and became its own chunk (with `max_tokens=60`, twelve ten-token sections produced twelve chunks instead of two full ones). Both measurements now go through `ChunkingOptions.measure()`, as the rest of the class already does. Character-mode behavior is unchanged.

## 0.27.17

### Fixes
Expand Down Expand Up @@ -73,7 +79,6 @@
- **`partition_doc()` and `partition_ppt()` no longer fail on a document whose name contains multi-byte characters.** `convert_office_doc()` decoded `soffice` stdout and stderr with a strict UTF-8 decode purely to log them and to check whether stdout was empty. LibreOffice echoes the input path using the console encoding, which on Windows is the locale codepage, so a document whose name or path contains multi-byte characters raised `UnicodeDecodeError` and aborted a conversion that would otherwise have succeeded. All three decode sites now go through one helper using `errors="backslashreplace"`, which keeps the message pure ASCII -- readable, still loggable by a handler using the locale codepage, and showing the offending bytes. Resolves #3652.

- **A stray processing instruction no longer crashes HTML partitioning.** `partition_html` (and formats that route through it, such as `.md`) raised `AttributeError: 'lxml.etree._ProcessingInstruction' object has no attribute 'is_phrasing'` when the HTML contained a processing-instruction node like a `<?xml ...?>` declaration. The parser now drops processing instructions at parse time, the same way it already drops comments.

## 0.27.7

### Fixes
Expand Down
19 changes: 19 additions & 0 deletions test_unstructured/chunking/conftest.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,19 @@
import pytest

from test_unstructured.unit_utils import FixtureRequest, Mock, method_mock
from unstructured.chunking.base import TokenCounter


@pytest.fixture()
def word_token_counter_(request: FixtureRequest) -> Mock:
"""Count one token per whitespace-delimited word.

Deterministic and offline, so a test can state exact token counts without depending on
tiktoken being installed or on the contents of a particular encoding.
"""
return method_mock(
request,
TokenCounter,
"count",
side_effect=lambda _, text: len(text.split()),
)
45 changes: 45 additions & 0 deletions test_unstructured/chunking/test_base.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,7 @@
from lxml.html import fragment_fromstring

import unstructured.chunking.base as chunking_base
from test_unstructured.unit_utils import Mock
from unstructured.chunking.base import (
ChunkingOptions,
PreChunk,
Expand Down Expand Up @@ -828,6 +829,50 @@ def it_knows_when_it_can_combine_itself_with_another_PreChunk_instance(

assert pre_chunk.can_combine(next_pre_chunk) is expected_value

@pytest.mark.parametrize(
("max_tokens", "combine_text_under_n_chars", "expected_value"),
[
# Will exactly fit:
# - text = 7 tokens < combine_text_under_n_chars
# - pre_chunk + separator + next_pre_chunk_text = 7 + 0 + 5 = 12 <= max_tokens
(12, 8, True),
# -- already exceeds the combine threshold, which is also counted in tokens --
(12, 7, False),
# -- would exceed the hard-max chunking-window threshold --
(11, 8, False),
],
)
def it_measures_in_tokens_when_it_can_combine_itself_with_another_PreChunk_instance(
self,
max_tokens: int,
combine_text_under_n_chars: int,
expected_value: bool,
word_token_counter_: Mock,
):
"""Both thresholds are token counts in token mode, not character counts.

Measuring the text in characters here made `.can_combine()` answer `False` for every
short pre-chunk, because even a handful of tokens is more characters than the token
budget. `chunk_by_title()` then emitted one chunk per section.
"""
opts = ChunkingOptions(
max_tokens=max_tokens,
tokenizer="cl100k_base",
combine_text_under_n_chars=combine_text_under_n_chars,
)
pre_chunk = PreChunk(
[Text("Lorem ipsum dolor sit amet consectetur adipiscing.")], # -- 7 tokens, 50 chars
overlap_prefix="",
opts=opts,
)
next_pre_chunk = PreChunk(
[Text("In rhoncus sum sed lectus.")], # -- 5 tokens, 26 chars
overlap_prefix="",
opts=opts,
)

assert pre_chunk.can_combine(next_pre_chunk) is expected_value

def it_does_not_combine_when_either_pre_chunk_contains_a_table(self):
opts = ChunkingOptions(max_characters=500, combine_text_under_n_chars=500)
text_pre_chunk = PreChunk([Text("hello")], overlap_prefix="", opts=opts)
Expand Down
40 changes: 40 additions & 0 deletions test_unstructured/chunking/test_title.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,7 @@

from test_unstructured.unit_utils import FixtureRequest, Mock, function_mock, input_path
from unstructured.chunking.base import CHUNK_MULTI_PAGE_DEFAULT
from unstructured.chunking.basic import chunk_elements
from unstructured.chunking.title import _ByTitleChunkingOptions, chunk_by_title
from unstructured.documents.coordinates import CoordinateSystem
from unstructured.documents.elements import (
Expand Down Expand Up @@ -888,3 +889,42 @@ def it_applies_token_based_overlap_in_split_chunks(self, _tiktoken_installed: No
for chunk in chunks:
token_count = len(enc.encode(chunk.text))
assert token_count <= 12, f"Chunk exceeded token limit: {token_count} tokens"

def it_fills_the_token_budget_when_combining_short_sections(self, word_token_counter_: Mock):
"""Short sections combine up to `max_tokens`, like `chunk_elements()` does.

`combine_text_under_n_chars` defaults to `max_tokens` in token mode, so a section had to
be under that many *tokens* to be combined. Measuring it in characters instead put every
section over the threshold and suppressed combining altogether, leaving one chunk per
section at a fraction of the requested chunk size.
"""
elements: list[Element] = []
for idx in range(12):
# -- each section is 2 + 8 == 10 tokens, a sixth of the budget --
elements.append(Title(f"Section {idx}"))
elements.append(Text("alpha beta gamma delta epsilon zeta seven eight"))

chunks = chunk_by_title(elements, max_tokens=60, tokenizer="cl100k_base")

assert [len(chunk.text.split()) for chunk in chunks] == [60, 60]
# -- and the section boundaries cost nothing relative to ignoring them entirely --
assert [chunk.text for chunk in chunks] == [
chunk.text for chunk in chunk_elements(elements, max_tokens=60, tokenizer="cl100k_base")
]

def it_still_measures_the_combine_threshold_in_characters_in_character_mode(self):
"""Character mode is unaffected: `combine_text_under_n_chars` stays a character count."""
elements: list[Element] = [
Title("Alpha"),
Text("one two three four five"), # -- 5 tokens but 23 characters --
Title("Bravo"),
Text("six seven eight nine ten"),
]

chunks = chunk_by_title(elements, max_characters=200, combine_text_under_n_chars=20)

# -- 29 characters is over the 20-character threshold, so no combining --
assert [chunk.text for chunk in chunks] == [
"Alpha\n\none two three four five",
"Bravo\n\nsix seven eight nine ten",
]
2 changes: 1 addition & 1 deletion unstructured/__version__.py
Original file line number Diff line number Diff line change
@@ -1 +1 @@
__version__ = "0.27.17" # pragma: no cover
__version__ = "0.27.18" # pragma: no cover
6 changes: 4 additions & 2 deletions unstructured/chunking/base.py
Original file line number Diff line number Diff line change
Expand Up @@ -702,13 +702,15 @@ def can_combine(self, pre_chunk: PreChunk) -> bool:
self._elements, pre_chunk._elements
):
return False
if len(self._text) >= self._opts.combine_text_under_n_chars:
# -- both thresholds below are expressed in the configured units, so the text has to be
# -- sized in those units too; `measure()` is `len()` in character mode --
if self._opts.measure(self._text) >= self._opts.combine_text_under_n_chars:
return False
# -- avoid duplicating length computations by doing a trial-combine which is just as
# -- efficient and definitely more robust than hoping two different computations of combined
# -- length continue to get the same answer as the code evolves. Only possible because
# -- `.combine()` is non-mutating.
combined_len = len(self.combine(pre_chunk)._text)
combined_len = self._opts.measure(self.combine(pre_chunk)._text)

return combined_len <= self._opts.hard_max

Expand Down