Split a large paragraph that exceeds max_tokens.
(self, text: str, max_tokens: int)
| 185 | return chunks |
| 186 | |
| 187 | def _split_large_paragraph(self, text: str, max_tokens: int) -> List[str]: |
| 188 | """Split a large paragraph that exceeds max_tokens.""" |
| 189 | chunks = [] |
| 190 | |
| 191 | # Try to split by sentences (Chinese and English punctuation) |
| 192 | sentences = re.split(r'([。!?.!?]+)', text) |
| 193 | |
| 194 | # Recombine sentences with their punctuation |
| 195 | combined_sentences = [] |
| 196 | for i in range(0, len(sentences) - 1, 2): |
| 197 | if i + 1 < len(sentences): |
| 198 | combined_sentences.append(sentences[i] + sentences[i + 1]) |
| 199 | else: |
| 200 | combined_sentences.append(sentences[i]) |
| 201 | if len(sentences) % 2 == 1 and sentences[-1].strip(): |
| 202 | combined_sentences.append(sentences[-1]) |
| 203 | |
| 204 | current_chunk = "" |
| 205 | current_tokens = 0 |
| 206 | |
| 207 | for sent in combined_sentences: |
| 208 | sent_tokens = self._llm_client.count_tokens(sent) |
| 209 | |
| 210 | if sent_tokens > max_tokens: |
| 211 | # Sentence itself is too long, split by character count |
| 212 | if current_chunk.strip(): |
| 213 | chunks.append(current_chunk.strip()) |
| 214 | current_chunk = "" |
| 215 | current_tokens = 0 |
| 216 | |
| 217 | # Estimate chars per token (conservative for Chinese) |
| 218 | chars_per_chunk = int(max_tokens * 1.5) |
| 219 | for i in range(0, len(sent), chars_per_chunk): |
| 220 | chunks.append(sent[i:i + chars_per_chunk]) |
| 221 | elif current_tokens + sent_tokens > max_tokens: |
| 222 | if current_chunk.strip(): |
| 223 | chunks.append(current_chunk.strip()) |
| 224 | current_chunk = sent |
| 225 | current_tokens = sent_tokens |
| 226 | else: |
| 227 | current_chunk += sent |
| 228 | current_tokens += sent_tokens |
| 229 | |
| 230 | if current_chunk.strip(): |
| 231 | chunks.append(current_chunk.strip()) |
| 232 | |
| 233 | return chunks |
| 234 | |
| 235 | def _truncate_to_token_limit(self, text: str, max_tokens: int) -> str: |
| 236 | """Truncate text to fit within token limit using _llm_client tokenizer.""" |
no test coverage detected