Validate that extracted content is the actual guidance document. Returns (is_valid, reason).
(
markdown: str, expected_title: str
)
| 665 | def _near_title_match(norm_expected: str, norm_first: str) -> bool: |
| 666 | """True when expected title is nearly the document heading (small inserts).""" |
| 667 | if not norm_expected or len(norm_expected) < 10: |
| 668 | return False |
| 669 | # Prefer matching against an early heading-like span containing 指导原则 |
| 670 | heading = "" |
| 671 | for m in re.finditer(r".{0,80}指导原则", norm_first[:1200]): |
| 672 | cand = m.group(0) |
| 673 | # Drop leading markdown / 附件N noise that breaks containment checks |
| 674 | cand = re.sub(r'^[#\s]*附件\d*', '', cand) |
| 675 | cand = re.sub(r'^#+', '', cand) |
| 676 | if len(cand) >= 10: |
| 677 | heading = cand |
| 678 | break |
| 679 | # Also allow full early window containment (wrapped titles with : /) |
| 680 | if norm_expected in norm_first[:400]: |
| 681 | return True |
| 682 | target = heading or norm_first[:120] |
| 683 | if not target: |
| 684 | return False |
| 685 | if norm_expected in target or target in norm_expected: |
| 686 | return True |
| 687 | shorter, longer = ( |
| 688 | (norm_expected, target) |
| 689 | if len(norm_expected) <= len(target) |
no test coverage detected