Skip to content

ontocast.tool.chunk.util

Attributes

SENTENCE_SPLIT_REGEX = '(?:\\n\\s*\\n+)|(?<=[.!?])\\s+(?=[A-Z][a-z])' module-attribute

__all__ = ['SENTENCE_SPLIT_REGEX', 'SemanticChunker', 'split_proposition_windows'] module-attribute

Classes

SemanticChunker

Bases: BaseDocumentTransformer

Source code in ontocast/tool/chunk/util.py
class SemanticChunker(BaseDocumentTransformer):
    def __init__(
        self,
        embeddings: Embeddings,
        chunk_config: ChunkConfig,
        sentence_split_regex: str,
    ):
        """Initialize SemanticChunker.

        Args:
            embeddings: Embeddings model for generating sentence embeddings.
            chunk_config: Chunking configuration containing min_size, max_size, etc.
            sentence_split_regex: Regular expression pattern for splitting text into sentences.
        """
        self.embeddings = embeddings
        self.chunk_config = chunk_config
        self.min_size = chunk_config.min_size
        self.max_size = chunk_config.max_size
        self.sentence_split_regex = sentence_split_regex

    def _build_sentence_windows(
        self,
        sentences: List[str],
        window_size: int = _SENTENCE_WINDOW_SIZE,
    ) -> List[str]:
        if len(sentences) <= window_size:
            return [" ".join(sentences)]

        windows = []
        for i in range(len(sentences)):
            start = max(0, i - window_size // 2)
            end = min(len(sentences), start + window_size)
            window = " ".join(sentences[start:end])
            windows.append(window)

        return windows

    def _get_embeddings(self, sentences: List[str]) -> np.ndarray:
        """Embeds sentences directly without buffering.

        Since we cluster all sentences together, the clustering algorithm
        naturally captures semantic relationships without needing context windows.
        """
        return np.array(self.embeddings.embed_documents(sentences))

    def _cluster_sentences(
        self, vectors: np.ndarray, sentences: List[str]
    ) -> np.ndarray:
        """Pipeline: PCA -> UMAP -> HDBSCAN with parameters favoring more clusters.

        Uses HDBSCAN hyperparameters tuned to create more clusters, which helps
        ensure chunks respect max_size constraints. Large clusters will be split
        post-processing.

        Deterministic: the reduction is seeded, so the same vectors always get
        the same labels, in any process.

        Args:
            vectors: Embedding vectors for sentences.
            sentences: Original sentence texts for length validation.

        Returns:
            Cluster labels for each sentence.
        """
        # 1. PCA to reduce noise
        pca_dims = min(vectors.shape[0] - 1, 50)
        if pca_dims > 1:
            # Seeded: svd_solver="auto" picks the randomized solver on large inputs.
            vectors = PCA(
                n_components=pca_dims, random_state=_CLUSTER_RANDOM_STATE
            ).fit_transform(vectors)

        # 2. UMAP to 5 dimensions
        # n_neighbors=2 captures very local structure for chunking. That graph
        # is always disconnected, and the default spectral init lays its
        # components out differently on every run even when seeded, so the
        # init is PCA. umap forces n_jobs=1 once seeded; passing it avoids the
        # override warning.
        reducer = UMAP(
            n_components=5,
            n_neighbors=2,
            min_dist=0.0,
            metric="cosine",
            init="pca",
            random_state=_CLUSTER_RANDOM_STATE,
            n_jobs=1,
        )
        reduced_vectors = reducer.fit_transform(vectors)

        # 3. HDBSCAN with parameters favoring more clusters
        # Calculate optimal min_cluster_size based on max_size constraint
        # We want clusters small enough that they can be combined without exceeding max_size
        if len(sentences) > 0:
            avg_sentence_len = sum(len(s) for s in sentences) / len(sentences)
            # Target: clusters should be small enough that 2-3 clusters can fit in max_size
            # This encourages more, smaller clusters
            target_cluster_size = max(2, int(self.max_size / (avg_sentence_len * 2.5)))
            min_cluster_size = min(target_cluster_size, len(sentences) // 3, 10)
            min_cluster_size = max(2, min_cluster_size)  # At least 2, at most 10
        else:
            min_cluster_size = 2

        # Use cluster_selection_epsilon to encourage more splits
        # Higher epsilon = more aggressive splitting = more clusters
        # We use a small epsilon (0.1-0.3) to encourage splits while maintaining semantics
        clusterer = HDBSCAN(
            min_cluster_size=min_cluster_size,
            min_samples=1,  # Lower min_samples = more clusters
            metric="euclidean",
            cluster_selection_epsilon=0.1,  # Encourage more splits
            cluster_selection_method="eom",  # Excess of Mass method
        )
        labels = clusterer.fit_predict(reduced_vectors)

        return labels

    def split_text(self, text: str) -> List[str]:
        # Atomic split into sentences - chunks must contain whole sentences
        # Use capturing groups to preserve delimiters
        # Wrap the regex in a capturing group so delimiters are included in split result
        pattern_with_capture = f"({self.sentence_split_regex})"
        parts = re.split(pattern_with_capture, text)

        # Reconstruct sentences with their following delimiters
        # parts alternates: [text1, delimiter1, text2, delimiter2, ..., textN]
        # Handle case where text starts with delimiter (parts[0] empty)
        sentences = []
        delimiters = []  # Track delimiter after each sentence

        # Skip leading empty part if text starts with delimiter
        start_idx = 1 if parts and not parts[0].strip() else 0

        i = start_idx
        while i < len(parts):
            if i % 2 == start_idx % 2:  # Text parts (same parity as start)
                text_part = parts[i].strip()
                if text_part:  # Non-empty text
                    sentences.append(parts[i])  # Keep original (with whitespace)
                    # Get the delimiter that follows (if any)
                    if i + 1 < len(parts):
                        delimiters.append(parts[i + 1])
                    else:
                        delimiters.append("")  # No delimiter after last sentence
            i += 1

        # Filter out empty sentences
        if not sentences:
            return [text] if text.strip() else []

        if len(sentences) <= 1:
            # If single sentence, return it even if it exceeds max_size
            # (we can't split sentences, so we must keep it whole)
            return sentences

        if len(sentences) <= _SENTENCE_WINDOW_SIZE:
            # One window for all sentences: nothing to cluster, pack by size.
            labels = np.zeros(len(sentences), dtype=int)
        else:
            windows = self._build_sentence_windows(
                sentences, window_size=_SENTENCE_WINDOW_SIZE
            )
            vectors = self._get_embeddings(windows)
            labels = self._cluster_sentences(vectors, sentences)

        # Process sentences in original order, grouping consecutive sentences
        # from the same cluster into chunks
        chunks = []
        i = 0
        while i < len(sentences):
            label = labels[i]

            # Collect consecutive sentences with the same label
            cluster_sentences = [sentences[i]]
            cluster_delimiters = [delimiters[i] if i < len(delimiters) else ""]
            i += 1

            while i < len(sentences) and labels[i] == label:
                cluster_sentences.append(sentences[i])
                cluster_delimiters.append(delimiters[i] if i < len(delimiters) else "")
                i += 1

            # Process this cluster
            if label == -1:
                # Noise cluster: each sentence becomes its own chunk
                for idx, sentence in enumerate(cluster_sentences):
                    chunk = sentence
                    if idx < len(cluster_delimiters) and cluster_delimiters[idx]:
                        chunk += cluster_delimiters[idx]
                    chunks.append(chunk)
            else:
                # Regular cluster: group sentences respecting max_size
                cluster_len = sum(len(s) for s in cluster_sentences)
                delimiter_len = sum(len(d) for d in cluster_delimiters)
                total_cluster_len = cluster_len + delimiter_len

                if total_cluster_len <= self.max_size:
                    # Cluster fits in one chunk
                    chunk_parts = []
                    for j, sentence in enumerate(cluster_sentences):
                        chunk_parts.append(sentence)
                        if j < len(cluster_delimiters) and cluster_delimiters[j]:
                            chunk_parts.append(cluster_delimiters[j])
                    chunks.append("".join(chunk_parts))
                else:
                    # Split cluster into multiple chunks
                    current_chunk = []
                    current_delims = []
                    current_len = 0

                    for j, sentence in enumerate(cluster_sentences):
                        sentence_len = len(sentence)
                        delim = (
                            cluster_delimiters[j] if j < len(cluster_delimiters) else ""
                        )
                        delim_len = len(delim)

                        if current_len + sentence_len + delim_len > self.max_size:
                            # Current chunk is full
                            if current_chunk:
                                chunk_parts = []
                                for k, s in enumerate(current_chunk):
                                    chunk_parts.append(s)
                                    if k < len(current_delims) and current_delims[k]:
                                        chunk_parts.append(current_delims[k])
                                chunks.append("".join(chunk_parts))
                            current_chunk = [sentence]
                            current_delims = [delim]
                            current_len = sentence_len + delim_len
                        else:
                            current_chunk.append(sentence)
                            current_delims.append(delim)
                            current_len += sentence_len + delim_len

                    # Add remaining chunk
                    if current_chunk:
                        chunk_parts = []
                        for k, s in enumerate(current_chunk):
                            chunk_parts.append(s)
                            if k < len(current_delims) and current_delims[k]:
                                chunk_parts.append(current_delims[k])
                        chunks.append("".join(chunk_parts))

        return merge_small_parts(
            chunks,
            self.min_size,
            self.max_size,
            separator="",
        )

    def transform_documents(
        self, documents: Sequence[Document], **kwargs: Any
    ) -> Sequence[Document]:
        return self.split_documents(list(documents))

    def create_documents(
        self, texts: List[str], metadatas: List[dict] | None = None
    ) -> List[Document]:
        _metadatas = metadatas or [{}] * len(texts)
        documents = []
        for i, text in enumerate(texts):
            for chunk in self.split_text(text):
                metadata = copy.deepcopy(_metadatas[i])
                documents.append(Document(page_content=chunk, metadata=metadata))
        return documents

    def split_documents(self, documents: Iterable[Document]) -> List[Document]:
        texts = []
        metadatas = []
        for doc in documents:
            texts.append(doc.page_content)
            metadatas.append(doc.metadata)
        return self.create_documents(texts, metadatas)

Attributes

chunk_config = chunk_config instance-attribute
embeddings = embeddings instance-attribute
max_size = chunk_config.max_size instance-attribute
min_size = chunk_config.min_size instance-attribute
sentence_split_regex = sentence_split_regex instance-attribute

Methods:

__init__(embeddings, chunk_config, sentence_split_regex)

Initialize SemanticChunker.

Parameters:

Name Type Description Default
embeddings Embeddings

Embeddings model for generating sentence embeddings.

required
chunk_config ChunkConfig

Chunking configuration containing min_size, max_size, etc.

required
sentence_split_regex str

Regular expression pattern for splitting text into sentences.

required
Source code in ontocast/tool/chunk/util.py
def __init__(
    self,
    embeddings: Embeddings,
    chunk_config: ChunkConfig,
    sentence_split_regex: str,
):
    """Initialize SemanticChunker.

    Args:
        embeddings: Embeddings model for generating sentence embeddings.
        chunk_config: Chunking configuration containing min_size, max_size, etc.
        sentence_split_regex: Regular expression pattern for splitting text into sentences.
    """
    self.embeddings = embeddings
    self.chunk_config = chunk_config
    self.min_size = chunk_config.min_size
    self.max_size = chunk_config.max_size
    self.sentence_split_regex = sentence_split_regex
create_documents(texts, metadatas=None)
Source code in ontocast/tool/chunk/util.py
def create_documents(
    self, texts: List[str], metadatas: List[dict] | None = None
) -> List[Document]:
    _metadatas = metadatas or [{}] * len(texts)
    documents = []
    for i, text in enumerate(texts):
        for chunk in self.split_text(text):
            metadata = copy.deepcopy(_metadatas[i])
            documents.append(Document(page_content=chunk, metadata=metadata))
    return documents
split_documents(documents)
Source code in ontocast/tool/chunk/util.py
def split_documents(self, documents: Iterable[Document]) -> List[Document]:
    texts = []
    metadatas = []
    for doc in documents:
        texts.append(doc.page_content)
        metadatas.append(doc.metadata)
    return self.create_documents(texts, metadatas)
split_text(text)
Source code in ontocast/tool/chunk/util.py
def split_text(self, text: str) -> List[str]:
    # Atomic split into sentences - chunks must contain whole sentences
    # Use capturing groups to preserve delimiters
    # Wrap the regex in a capturing group so delimiters are included in split result
    pattern_with_capture = f"({self.sentence_split_regex})"
    parts = re.split(pattern_with_capture, text)

    # Reconstruct sentences with their following delimiters
    # parts alternates: [text1, delimiter1, text2, delimiter2, ..., textN]
    # Handle case where text starts with delimiter (parts[0] empty)
    sentences = []
    delimiters = []  # Track delimiter after each sentence

    # Skip leading empty part if text starts with delimiter
    start_idx = 1 if parts and not parts[0].strip() else 0

    i = start_idx
    while i < len(parts):
        if i % 2 == start_idx % 2:  # Text parts (same parity as start)
            text_part = parts[i].strip()
            if text_part:  # Non-empty text
                sentences.append(parts[i])  # Keep original (with whitespace)
                # Get the delimiter that follows (if any)
                if i + 1 < len(parts):
                    delimiters.append(parts[i + 1])
                else:
                    delimiters.append("")  # No delimiter after last sentence
        i += 1

    # Filter out empty sentences
    if not sentences:
        return [text] if text.strip() else []

    if len(sentences) <= 1:
        # If single sentence, return it even if it exceeds max_size
        # (we can't split sentences, so we must keep it whole)
        return sentences

    if len(sentences) <= _SENTENCE_WINDOW_SIZE:
        # One window for all sentences: nothing to cluster, pack by size.
        labels = np.zeros(len(sentences), dtype=int)
    else:
        windows = self._build_sentence_windows(
            sentences, window_size=_SENTENCE_WINDOW_SIZE
        )
        vectors = self._get_embeddings(windows)
        labels = self._cluster_sentences(vectors, sentences)

    # Process sentences in original order, grouping consecutive sentences
    # from the same cluster into chunks
    chunks = []
    i = 0
    while i < len(sentences):
        label = labels[i]

        # Collect consecutive sentences with the same label
        cluster_sentences = [sentences[i]]
        cluster_delimiters = [delimiters[i] if i < len(delimiters) else ""]
        i += 1

        while i < len(sentences) and labels[i] == label:
            cluster_sentences.append(sentences[i])
            cluster_delimiters.append(delimiters[i] if i < len(delimiters) else "")
            i += 1

        # Process this cluster
        if label == -1:
            # Noise cluster: each sentence becomes its own chunk
            for idx, sentence in enumerate(cluster_sentences):
                chunk = sentence
                if idx < len(cluster_delimiters) and cluster_delimiters[idx]:
                    chunk += cluster_delimiters[idx]
                chunks.append(chunk)
        else:
            # Regular cluster: group sentences respecting max_size
            cluster_len = sum(len(s) for s in cluster_sentences)
            delimiter_len = sum(len(d) for d in cluster_delimiters)
            total_cluster_len = cluster_len + delimiter_len

            if total_cluster_len <= self.max_size:
                # Cluster fits in one chunk
                chunk_parts = []
                for j, sentence in enumerate(cluster_sentences):
                    chunk_parts.append(sentence)
                    if j < len(cluster_delimiters) and cluster_delimiters[j]:
                        chunk_parts.append(cluster_delimiters[j])
                chunks.append("".join(chunk_parts))
            else:
                # Split cluster into multiple chunks
                current_chunk = []
                current_delims = []
                current_len = 0

                for j, sentence in enumerate(cluster_sentences):
                    sentence_len = len(sentence)
                    delim = (
                        cluster_delimiters[j] if j < len(cluster_delimiters) else ""
                    )
                    delim_len = len(delim)

                    if current_len + sentence_len + delim_len > self.max_size:
                        # Current chunk is full
                        if current_chunk:
                            chunk_parts = []
                            for k, s in enumerate(current_chunk):
                                chunk_parts.append(s)
                                if k < len(current_delims) and current_delims[k]:
                                    chunk_parts.append(current_delims[k])
                            chunks.append("".join(chunk_parts))
                        current_chunk = [sentence]
                        current_delims = [delim]
                        current_len = sentence_len + delim_len
                    else:
                        current_chunk.append(sentence)
                        current_delims.append(delim)
                        current_len += sentence_len + delim_len

                # Add remaining chunk
                if current_chunk:
                    chunk_parts = []
                    for k, s in enumerate(current_chunk):
                        chunk_parts.append(s)
                        if k < len(current_delims) and current_delims[k]:
                            chunk_parts.append(current_delims[k])
                    chunks.append("".join(chunk_parts))

    return merge_small_parts(
        chunks,
        self.min_size,
        self.max_size,
        separator="",
    )
transform_documents(documents, **kwargs)
Source code in ontocast/tool/chunk/util.py
def transform_documents(
    self, documents: Sequence[Document], **kwargs: Any
) -> Sequence[Document]:
    return self.split_documents(list(documents))

Functions:

split_proposition_windows(text, max_sentences=2, max_windows=16, stride=None, max_chars=None, max_tokens=None, token_counter=None, abbreviation_aware=False, measurement_aware=False, overlap=0.0)

Split text into short proposition-like windows for retrieval.

Parameters:

Name Type Description Default
text str

Passage to split.

required
max_sentences int

Sentences per window. Not applied when max_chars or max_tokens is set: the three are alternative ways of saying how much text one query carries, and honouring more than one means the tighter one silently wins.

2
max_windows int

Ceiling on the windows returned; over it, windows are sampled evenly across the passage rather than truncated, so the tail still contributes a query.

16
stride int | None

Sentences advanced between windows. None (default) strides by the window's own length, giving contiguous, disjoint windows. A smaller stride overlaps them, which is what lets a statement spanning a window boundary appear whole in some window -- with disjoint windows it appears in none, and neither half retrieves what the pair together names. Sentence-granular, so it says nothing about how much text is repeated; overlap is the fractional form the budget modes take.

None
max_chars int | None

Characters per window. None (default) bounds windows by sentence count instead, which is the historical behaviour and is reproduced exactly.

A sentence count is a poor bound on how much text a query carries. Two sentences of technical prose range over an order of magnitude in length, and the encoder truncates on tokens, so a sentence-bounded window can silently lose its tail. The splitter also has no abbreviation handling and breaks on every ., so "J. Phys. Chem. Lett." is four "sentences" and a two-sentence window over a citation carries nothing retrievable at all.

A character budget addresses both ends: it caps the long windows that truncate, and it coalesces short fragments, because it keeps taking sentences until the budget is reached. A single sentence longer than the budget is emitted whole rather than cut -- the encoder truncates it either way, and cutting first only loses more.

None
max_tokens int | None

Encoder word pieces per window, measured with token_counter. None (default) leaves the other bounds in charge; when set it replaces both of them, and characters stop being consulted at all.

This is the only bound that equalises what a query carries, because it is the unit the encoder itself counts in: characters per token drift with notation, so a character budget that fits one passage truncates the next. Set below the encoder's sequence limit, it makes truncation impossible by construction -- which a character budget can only approximate. Unlike the character budget it does cut a sentence that alone exceeds the budget, at a whitespace boundary, because a sentence emitted whole would be truncated by the encoder anyway and the tail would then reach no lane at all.

None
token_counter TokenCounter | None

Word pieces per text, typically EmbeddingTool.token_lengths. Required by max_tokens; when it is absent or reports None, the budget degrades to a character approximation (FALLBACK_CHARS_PER_TOKEN) and says so once, rather than silently reverting to the sentence bound.

None
abbreviation_aware bool

Merge fragments the period-splitter created inside an abbreviation, an initial or a citation run. A prerequisite for any equal-size packing rather than a candidate of its own: without it the packer's atoms include "Chem. Lett.", and a window bound counted in those atoms is not counting sentences. General English and bibliographic shapes only -- see :data:ABBREVIATIONS.

False
measurement_aware bool

Forbid a break between a number and the unit it is written with, or inside a range, using the number/unit shapes in :mod:ontocast.util.measurement_lexicon. Only binds where a cut inside a sentence is possible, which today means max_tokens with a sentence over budget: a window ending on "a red shift of ~10" retrieves nothing that "~10 meV" would.

False
overlap float

Fraction of a window repeated at the start of the next one, for the budget modes. 0.0 (default) leaves windows disjoint. The fractional form exists because stride counts sentences, which under a budget says nothing about how much text is shared. Overlap multiplies queries, so it is paid for per window and, once max_windows binds, in coverage elsewhere in the passage.

0.0

Returns:

Type Description
list[str]

list[str]: Windows in document order.

Source code in ontocast/tool/chunk/proposition.py
def split_proposition_windows(
    text: str,
    max_sentences: int = 2,
    max_windows: int = 16,
    stride: int | None = None,
    max_chars: int | None = None,
    max_tokens: int | None = None,
    token_counter: TokenCounter | None = None,
    abbreviation_aware: bool = False,
    measurement_aware: bool = False,
    overlap: float = 0.0,
) -> list[str]:
    """Split text into short proposition-like windows for retrieval.

    Args:
        text: Passage to split.
        max_sentences: Sentences per window. Not applied when ``max_chars`` or
            ``max_tokens`` is set: the three are alternative ways of saying how
            much text one query carries, and honouring more than one means the
            tighter one silently wins.
        max_windows: Ceiling on the windows returned; over it, windows are sampled
            evenly across the passage rather than truncated, so the tail still
            contributes a query.
        stride: Sentences advanced between windows. ``None`` (default) strides by
            the window's own length, giving contiguous, disjoint windows. A smaller
            stride overlaps them, which is what lets a statement spanning a window
            boundary appear whole in some window -- with disjoint windows it appears
            in none, and neither half retrieves what the pair together names.
            Sentence-granular, so it says nothing about how much text is repeated;
            ``overlap`` is the fractional form the budget modes take.
        max_chars: Characters per window. ``None`` (default) bounds windows by
            sentence count instead, which is the historical behaviour and is
            reproduced exactly.

            A sentence count is a poor bound on how much text a query carries. Two
            sentences of technical prose range over an order of magnitude in
            length, and the encoder truncates on *tokens*, so a sentence-bounded
            window can silently lose its tail. The splitter also has no
            abbreviation handling and breaks on every ``.``, so "J. Phys. Chem.
            Lett." is four "sentences" and a two-sentence window over a citation
            carries nothing retrievable at all.

            A character budget addresses both ends: it caps the long windows that
            truncate, and it *coalesces* short fragments, because it keeps taking
            sentences until the budget is reached. A single sentence longer than
            the budget is emitted whole rather than cut -- the encoder truncates it
            either way, and cutting first only loses more.
        max_tokens: Encoder word pieces per window, measured with ``token_counter``.
            ``None`` (default) leaves the other bounds in charge; when set it
            replaces both of them, and characters stop being consulted at all.

            This is the only bound that equalises what a query carries, because it
            is the unit the encoder itself counts in: characters per token drift
            with notation, so a character budget that fits one passage truncates
            the next. Set below the encoder's sequence limit, it makes truncation
            impossible by construction -- which a character budget can only
            approximate. Unlike the character budget it *does* cut a sentence that
            alone exceeds the budget, at a whitespace boundary, because a sentence
            emitted whole would be truncated by the encoder anyway and the tail
            would then reach no lane at all.
        token_counter: Word pieces per text, typically
            ``EmbeddingTool.token_lengths``. Required by ``max_tokens``; when it is
            absent or reports ``None``, the budget degrades to a character
            approximation (``FALLBACK_CHARS_PER_TOKEN``) and says so once, rather
            than silently reverting to the sentence bound.
        abbreviation_aware: Merge fragments the period-splitter created inside an
            abbreviation, an initial or a citation run. A prerequisite for any
            equal-size packing rather than a candidate of its own: without it the
            packer's atoms include "Chem. Lett.", and a window bound counted in
            those atoms is not counting sentences. General English and
            bibliographic shapes only -- see :data:`ABBREVIATIONS`.
        measurement_aware: Forbid a break between a number and the unit it is
            written with, or inside a range, using the number/unit *shapes* in
            :mod:`ontocast.util.measurement_lexicon`. Only binds where a cut
            inside a sentence is possible, which today means ``max_tokens`` with a
            sentence over budget: a window ending on "a red shift of ~10" retrieves
            nothing that "~10 meV" would.
        overlap: Fraction of a window repeated at the start of the next one, for
            the budget modes. ``0.0`` (default) leaves windows disjoint. The
            fractional form exists because ``stride`` counts sentences, which under
            a budget says nothing about how much text is shared. Overlap multiplies
            queries, so it is paid for per window and, once ``max_windows`` binds,
            in coverage elsewhere in the passage.

    Returns:
        list[str]: Windows in document order.
    """
    cleaned = text.strip()
    if not cleaned:
        return []
    if max_sentences <= 0:
        raise ValueError("max_sentences must be >= 1")
    if max_windows <= 0:
        raise ValueError("max_windows must be >= 1")
    if max_chars is not None and max_chars <= 0:
        raise ValueError("max_chars must be >= 1")
    if max_tokens is not None and max_tokens <= 0:
        raise ValueError("max_tokens must be >= 1")
    if not 0.0 <= overlap < 1.0:
        raise ValueError("overlap must be in [0, 1)")
    step = max_sentences if stride is None else stride
    if step <= 0:
        raise ValueError("stride must be >= 1")

    # Keep this splitter lightweight and deterministic.
    sentence_parts = [
        part.strip() for part in _SENTENCE_SPLIT.split(cleaned) if part.strip()
    ]
    if abbreviation_aware:
        sentence_parts = _merge_abbreviations(sentence_parts)
    if not sentence_parts:
        return [cleaned[:1000]] if cleaned else []

    pieces = _windows(
        sentence_parts,
        max_sentences=max_sentences,
        step=step,
        stride=stride,
        max_chars=max_chars,
        max_tokens=max_tokens,
        token_counter=token_counter,
        measurement_aware=measurement_aware,
        overlap=overlap,
    )

    windows: list[str] = []
    seen: set[str] = set()
    for window in pieces:
        # A stride below the window size makes the final windows suffixes of one
        # another once the tail is shorter than a full window; emitting those twice
        # would pay for an identical query and skew the fusion ranks it feeds.
        if window and window not in seen:
            seen.add(window)
            windows.append(window)

    if len(windows) > max_windows:
        # Sample evenly across the text rather than keeping the first ``max_windows``.
        # Truncating dropped the tail of a long chunk entirely, so its closing sections
        # never contributed a retrieval query at all. Positions span both endpoints, so
        # the final window is always represented.
        if max_windows == 1:
            windows = windows[:1]
        else:
            last = len(windows) - 1
            picked = {
                round(position * last / (max_windows - 1))
                for position in range(max_windows)
            }
            windows = [windows[index] for index in sorted(picked)]

    return windows or [cleaned[:1000]]