Skip to content

chunking

chunking

Document chunking with configurable size and overlap.

Splits text into fixed-size chunks (measured in whitespace-split tokens) with a configurable overlap. Paragraph boundaries are respected when they fall within the chunk window.

Classes

ChunkConfig dataclass

ChunkConfig(chunk_size: int = 512, chunk_overlap: int = 64, min_chunk_size: int = 50)

Parameters controlling the chunking strategy.

Functions
validate
validate() -> None

Reject values that cannot produce a finite, well-formed chunk stream.

Source code in src/openjarvis/tools/storage/chunking.py
def validate(self) -> None:
    """Reject values that cannot produce a finite, well-formed chunk stream."""
    if (
        isinstance(self.chunk_size, bool)
        or not isinstance(self.chunk_size, int)
        or self.chunk_size < 1
    ):
        raise ValueError("chunk_size must be an integer >= 1")
    if (
        isinstance(self.chunk_overlap, bool)
        or not isinstance(self.chunk_overlap, int)
        or self.chunk_overlap < 0
    ):
        raise ValueError("chunk_overlap must be an integer >= 0")

Chunk dataclass

Chunk(content: str, source: str = '', offset: int = 0, index: int = 0, metadata: Dict[str, Any] = dict())

A single chunk produced by the chunking pipeline.

Functions

chunk_text

chunk_text(text: str, *, source: str = '', config: Optional[ChunkConfig] = None) -> List[Chunk]

Split text into chunks respecting paragraph boundaries.

PARAMETER DESCRIPTION
text

The full document text.

TYPE: str

source

Originating filename or identifier.

TYPE: str DEFAULT: ''

config

Chunking parameters (uses defaults if None).

TYPE: Optional[ChunkConfig] DEFAULT: None

RETURNS DESCRIPTION
List of :class:`Chunk` objects, in order.
Source code in src/openjarvis/tools/storage/chunking.py
def chunk_text(
    text: str,
    *,
    source: str = "",
    config: Optional[ChunkConfig] = None,
) -> List[Chunk]:
    """Split *text* into chunks respecting paragraph boundaries.

    Parameters
    ----------
    text:
        The full document text.
    source:
        Originating filename or identifier.
    config:
        Chunking parameters (uses defaults if ``None``).

    Returns
    -------
    List of :class:`Chunk` objects, in order.
    """
    cfg = config or ChunkConfig()
    # ``ChunkConfig`` is intentionally mutable for API compatibility. Recheck
    # here so a caller cannot construct a valid config, mutate it, and enter a
    # non-progress loop during ingestion.
    cfg.validate()

    if not text or not text.strip():
        return []

    # Split into paragraphs (double newline)
    paragraphs = [p for p in text.split("\n\n") if p.strip()]

    chunks: List[Chunk] = []
    current_tokens: List[str] = []
    current_offset = 0
    chunk_start_offset = 0
    force_final_chunk = False

    def emit_current(consumed_offset: int) -> None:
        """Emit the buffer and retain a bounded overlap for the next chunk."""
        nonlocal current_tokens, chunk_start_offset, force_final_chunk

        chunks.append(
            Chunk(
                content=" ".join(current_tokens),
                source=source,
                offset=chunk_start_offset,
                index=len(chunks),
            )
        )

        # Even a malformed overlap >= chunk_size must leave room for forward
        # progress and may never make the next chunk exceed the hard bound.
        keep = min(cfg.chunk_overlap, max(0, len(current_tokens) - 1))
        current_tokens = current_tokens[-keep:] if keep else []
        chunk_start_offset = consumed_offset - keep
        force_final_chunk = False

    for para in paragraphs:
        para_tokens = para.split()
        would_overflow = len(current_tokens) + len(para_tokens) > cfg.chunk_size

        # Preserve a paragraph boundary when the buffered paragraph is large
        # enough to stand alone.  A sub-floor lead-in instead stays attached
        # to the following text; dropping it was the original #754 bug.
        preserve_sequence = force_final_chunk
        if current_tokens and would_overflow:
            if len(current_tokens) >= cfg.min_chunk_size:
                emit_current(current_offset)
            else:
                preserve_sequence = True

        para_index = 0
        while para_index < len(para_tokens):
            capacity = cfg.chunk_size - len(current_tokens)
            remaining = len(para_tokens) - para_index

            if remaining <= capacity:
                current_tokens.extend(para_tokens[para_index:])
                para_index = len(para_tokens)
                force_final_chunk = preserve_sequence
                break

            take = capacity
            remainder = remaining - take
            overlap = min(cfg.chunk_overlap, max(0, cfg.chunk_size - 1))

            # With no (or a small) configured overlap, a full window can
            # strand a sub-floor tail.  Rebalance the two windows when both
            # can satisfy the minimum; otherwise the preservation flag below
            # keeps the unavoidable short tail rather than losing content.
            tail_size = overlap + remainder
            if preserve_sequence and tail_size < cfg.min_chunk_size:
                shift = cfg.min_chunk_size - tail_size
                if len(current_tokens) + take - shift >= cfg.min_chunk_size:
                    take -= shift

            current_tokens.extend(para_tokens[para_index : para_index + take])
            para_index += take
            emit_current(current_offset + para_index)
            force_final_chunk = preserve_sequence

        current_offset += len(para_tokens)

    # The buffer contains new document content, even when it also retains an
    # overlap. The minimum guides boundary selection; it must not discard a
    # short final paragraph or the end of an oversized paragraph (#754).
    if current_tokens:
        chunks.append(
            Chunk(
                content=" ".join(current_tokens),
                source=source,
                offset=chunk_start_offset,
                index=len(chunks),
            )
        )

    return chunks