codekingpro/portable-devtools
115k
1import Diff from './base.js';
2function isSentenceEndPunct(char) {
3 return char == '.' || char == '!' || char == '?';
4}
5class SentenceDiff extends Diff {
6 tokenize(value) {
7 var _a;
8 // If in future we drop support for environments that don't support lookbehinds, we can replace
9 // this entire function with:
10 // return value.split(/(?<=[.!?])(\s+|$)/);
11 // but until then, for similar reasons to the trailingWs function in string.ts, we are forced
12 // to do this verbosely "by hand" instead of using a regex.
13 const result = [];
14 let tokenStartI = 0;
15 for (let i = 0; i < value.length; i++) {
16 if (i == value.length - 1) {
17 result.push(value.slice(tokenStartI));
18 break;
19 }
20 if (isSentenceEndPunct(value[i]) && value[i + 1].match(/\s/)) {
21 // We've hit a sentence break - i.e. a punctuation mark followed by whitespace.
22 // We now want to push TWO tokens to the result:
23 // 1. the sentence
24 result.push(value.slice(tokenStartI, i + 1));
25 // 2. the whitespace
26 i = tokenStartI = i + 1;
27 while ((_a = value[i + 1]) === null || _a === void 0 ? void 0 : _a.match(/\s/)) {
28 i++;
29 }
30 result.push(value.slice(tokenStartI, i + 1));
31 // Then the next token (a sentence) starts on the character after the whitespace.
32 // (It's okay if this is off the end of the string - then the outer loop will terminate
33 // here anyway.)
34 tokenStartI = i + 1;
35 }
36 }
37 return result;
38 }
39}
40export const sentenceDiff = new SentenceDiff();
41export function diffSentences(oldStr, newStr, options) {
42 return sentenceDiff.diff(oldStr, newStr, options);
43}
44 