MCPcopy Create free account
hub / github.com/AQ-MedAI/MedMemoryBench / _split_large_paragraph

Method _split_large_paragraph

methods/amem_agent.py:187–233  ·  view source on GitHub ↗

Split a large paragraph that exceeds max_tokens.

(self, text: str, max_tokens: int)

Source from the content-addressed store, hash-verified

185 return chunks
186
187 def _split_large_paragraph(self, text: str, max_tokens: int) -> List[str]:
188 """Split a large paragraph that exceeds max_tokens."""
189 chunks = []
190
191 # Try to split by sentences (Chinese and English punctuation)
192 sentences = re.split(r'([。!?.!?]+)', text)
193
194 # Recombine sentences with their punctuation
195 combined_sentences = []
196 for i in range(0, len(sentences) - 1, 2):
197 if i + 1 < len(sentences):
198 combined_sentences.append(sentences[i] + sentences[i + 1])
199 else:
200 combined_sentences.append(sentences[i])
201 if len(sentences) % 2 == 1 and sentences[-1].strip():
202 combined_sentences.append(sentences[-1])
203
204 current_chunk = ""
205 current_tokens = 0
206
207 for sent in combined_sentences:
208 sent_tokens = self._llm_client.count_tokens(sent)
209
210 if sent_tokens > max_tokens:
211 # Sentence itself is too long, split by character count
212 if current_chunk.strip():
213 chunks.append(current_chunk.strip())
214 current_chunk = ""
215 current_tokens = 0
216
217 # Estimate chars per token (conservative for Chinese)
218 chars_per_chunk = int(max_tokens * 1.5)
219 for i in range(0, len(sent), chars_per_chunk):
220 chunks.append(sent[i:i + chars_per_chunk])
221 elif current_tokens + sent_tokens > max_tokens:
222 if current_chunk.strip():
223 chunks.append(current_chunk.strip())
224 current_chunk = sent
225 current_tokens = sent_tokens
226 else:
227 current_chunk += sent
228 current_tokens += sent_tokens
229
230 if current_chunk.strip():
231 chunks.append(current_chunk.strip())
232
233 return chunks
234
235 def _truncate_to_token_limit(self, text: str, max_tokens: int) -> str:
236 """Truncate text to fit within token limit using _llm_client tokenizer."""

Callers 1

Calls 1

count_tokensMethod · 0.45

Tested by

no test coverage detected