Split text into chunks with optional overlap.
(
self,
text: str,
max_tokens: int,
overlap_tokens: int = 0
)
| 556 | return self._tokenizer.decode(tokens[:max_tokens]) |
| 557 | |
| 558 | def _split_text_into_chunks( |
| 559 | self, |
| 560 | text: str, |
| 561 | max_tokens: int, |
| 562 | overlap_tokens: int = 0 |
| 563 | ) -> List[str]: |
| 564 | """Split text into chunks with optional overlap.""" |
| 565 | if not text.strip(): |
| 566 | return [] |
| 567 | |
| 568 | tokens = self._tokenizer.encode(text) |
| 569 | if len(tokens) <= max_tokens: |
| 570 | return [text] |
| 571 | |
| 572 | chunks: List[str] = [] |
| 573 | start = 0 |
| 574 | step = max(1, max_tokens - overlap_tokens) |
| 575 | |
| 576 | while start < len(tokens): |
| 577 | end = min(start + max_tokens, len(tokens)) |
| 578 | chunk_tokens = tokens[start:end] |
| 579 | chunks.append(self._tokenizer.decode(chunk_tokens)) |
| 580 | |
| 581 | if end >= len(tokens): |
| 582 | break |
| 583 | start += step |
| 584 | |
| 585 | return chunks |
| 586 | |
| 587 | def memorize(self, text: str, **kwargs) -> MemoryBuildResult: |
| 588 | """Store text into MemRL memory using official build strategy. |