( html: string, stripHtml: boolean, caseFoldBool: boolean, removeCombiningCharacters_DoNotEnableByDefault: boolean, )
| 22 | https://www.npmjs.com/package/unidecode |
| 23 | */ |
| 24 | export function ftsNormalize( |
| 25 | html: string, |
| 26 | stripHtml: boolean, |
| 27 | caseFoldBool: boolean, |
| 28 | |
| 29 | removeCombiningCharacters_DoNotEnableByDefault: boolean, |
| 30 | ) { |
| 31 | // "The W3C Character Model for the World Wide Web 1.0: Normalization [CharNorm]... recommend using Normalization Form C for all content" https://www.unicode.org/reports/tr15/ |
| 32 | let r = html.normalize() // defaults to "NFC" |
| 33 | if (stripHtml) { |
| 34 | r = ftsNormalizeHelper(html).replace(/\s+/g, ' ').trim() |
| 35 | } |
| 36 | // Normalization precedes case folding https://www.w3.org/TR/charmod-norm/#PreNormalization |
| 37 | // 1. Perform Unicode normalization of the string to form NFD or form NFC. |
| 38 | // 2. Perform Unicode Full case folding of the resulting string. |
| 39 | if (caseFoldBool) { |
| 40 | r = caseFold(r) |
| 41 | } |
| 42 | // Do not enable this by default! |
| 43 | // "a process that attempts to remove accents from letters by decomposing the text and then |
| 44 | // removing all of the combining characters will break languages that rely on combining marks" https://www.w3.org/TR/charmod-norm/#additionalMatchTailoring |
| 45 | if (removeCombiningCharacters_DoNotEnableByDefault) { |
| 46 | // https://stackoverflow.com/a/37511463 also grep 7D669B43-CEA7-45E6-AB82-D2B18C20D633 |
| 47 | r = r.normalize('NFKD').replace(/[\u0300-\u036f]/g, '') |
| 48 | } |
| 49 | return r |
| 50 | } |
| 51 | |
| 52 | const toOneLineHelper = compile({ |
| 53 | selectors: [ |
no test coverage detected