Calculate similarity between two text blocks with line-by-line fuzzy matching. This function compares blocks line-by-line and averages the similarity scores, making it robust to minor variations in each line. Args: text1: First text block (list of lines) text2: Sec
(text1: List[str], text2: List[str])
| 308 | |
| 309 | |
| 310 | def _calculate_similarity(text1: List[str], text2: List[str]) -> float: |
| 311 | """ |
| 312 | Calculate similarity between two text blocks with line-by-line fuzzy matching. |
| 313 | |
| 314 | This function compares blocks line-by-line and averages the similarity scores, |
| 315 | making it robust to minor variations in each line. |
| 316 | |
| 317 | Args: |
| 318 | text1: First text block (list of lines) |
| 319 | text2: Second text block (list of lines) |
| 320 | |
| 321 | Returns: |
| 322 | Similarity ratio between 0.0 and 1.0 |
| 323 | """ |
| 324 | if len(text1) != len(text2): |
| 325 | # If different number of lines, fall back to overall similarity |
| 326 | norm1 = [_normalize_line(line) for line in text1] |
| 327 | norm2 = [_normalize_line(line) for line in text2] |
| 328 | matcher = SequenceMatcher(None, norm1, norm2) |
| 329 | return matcher.ratio() |
| 330 | |
| 331 | # Line-by-line fuzzy matching |
| 332 | total_score = 0.0 |
| 333 | for line1, line2 in zip(text1, text2): |
| 334 | norm1 = _normalize_line(line1) |
| 335 | norm2 = _normalize_line(line2) |
| 336 | matcher = SequenceMatcher(None, norm1, norm2) |
| 337 | total_score += matcher.ratio() |
| 338 | |
| 339 | # Return average similarity across all lines |
| 340 | return total_score / len(text1) if text1 else 0.0 |
| 341 | |
| 342 | |
| 343 | def _find_best_match( |
no test coverage detected