Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes15kdownloads
word.js282 linesDownload Raw Back to diff
1import Diff from './base.js';
2import { longestCommonPrefix, longestCommonSuffix, replacePrefix, replaceSuffix, removePrefix, removeSuffix, maximumOverlap, leadingWs, trailingWs, leadingAndTrailingWs, segment } from '../util/string.js';
3// Based on https://en.wikipedia.org/wiki/Latin_script_in_Unicode
4//
5// Chars/ranges counted as "word" characters by this regex are as follows:
6//
7// + U+00AD  Soft hyphen
8// + 00C0–00FF (letters with diacritics from the Latin-1 Supplement), except:
9//   - U+00D7  × Multiplication sign
10//   - U+00F7  ÷ Division sign
11// + Latin Extended-A, 0100–017F
12// + Latin Extended-B, 0180–024F
13// + IPA Extensions, 0250–02AF
14// + Spacing Modifier Letters, 02B0–02FF, except:
15//   - U+02C7  ˇ ˇ  Caron
16//   - U+02D8  ˘ ˘  Breve
17//   - U+02D9  ˙ ˙  Dot Above
18//   - U+02DA  ˚ ˚  Ring Above
19//   - U+02DB  ˛ ˛  Ogonek
20//   - U+02DC  ˜ ˜  Small Tilde
21//   - U+02DD  ˝ ˝  Double Acute Accent
22// + Latin Extended Additional, 1E00–1EFF
23const extendedWordChars = 'a-zA-Z0-9_\\u{AD}\\u{C0}-\\u{D6}\\u{D8}-\\u{F6}\\u{F8}-\\u{2C6}\\u{2C8}-\\u{2D7}\\u{2DE}-\\u{2FF}\\u{1E00}-\\u{1EFF}';
24// Each token is one of the following:
25// - A punctuation mark plus the surrounding whitespace
26// - A word plus the surrounding whitespace
27// - Pure whitespace (but only in the special case where the entire text
28//   is just whitespace)
29//
30// We have to include surrounding whitespace in the tokens because the two
31// alternative approaches produce horribly broken results:
32// * If we just discard the whitespace, we can't fully reproduce the original
33//   text from the sequence of tokens and any attempt to render the diff will
34//   get the whitespace wrong.
35// * If we have separate tokens for whitespace, then in a typical text every
36//   second token will be a single space character. But this often results in
37//   the optimal diff between two texts being a perverse one that preserves
38//   the spaces between words but deletes and reinserts actual common words.
39//   See https://github.com/kpdecker/jsdiff/issues/160#issuecomment-1866099640
40//   for an example.
41//
42// Keeping the surrounding whitespace of course has implications for .equals
43// and .join, not just .tokenize.
44// This regex does NOT fully implement the tokenization rules described above.
45// Instead, it gives runs of whitespace their own "token". The tokenize method
46// then handles stitching whitespace tokens onto adjacent word or punctuation
47// tokens.
48const tokenizeIncludingWhitespace = new RegExp(`[${extendedWordChars}]+|\\s+|[^${extendedWordChars}]`, 'ug');
49class WordDiff extends Diff {
50    equals(left, right, options) {
51        if (options.ignoreCase) {
52            left = left.toLowerCase();
53            right = right.toLowerCase();
54        }
55        return left.trim() === right.trim();
56    }
57    tokenize(value, options = {}) {
58        let parts;
59        if (options.intlSegmenter) {
60            const segmenter = options.intlSegmenter;
61            if (segmenter.resolvedOptions().granularity != 'word') {
62                throw new Error('The segmenter passed must have a granularity of "word"');
63            }
64            // We want `parts` to be an array whose elements alternate between being
65            // pure whitespace and being pure non-whitespace. This is ALMOST what the
66            // segments returned by a word-based Intl.Segmenter already look like,
67            // but not quite - see explanation in the docs of our custom segment()
68            // function.
69            parts = segment(value, segmenter);
70        }
71        else {
72            parts = value.match(tokenizeIncludingWhitespace) || [];
73        }
74        const tokens = [];
75        let prevPart = null;
76        parts.forEach(part => {
77            if ((/\s/).test(part)) {
78                if (prevPart == null) {
79                    tokens.push(part);
80                }
81                else {
82                    tokens.push(tokens.pop() + part);
83                }
84            }
85            else if (prevPart != null && (/\s/).test(prevPart)) {
86                if (tokens[tokens.length - 1] == prevPart) {
87                    tokens.push(tokens.pop() + part);
88                }
89                else {
90                    tokens.push(prevPart + part);
91                }
92            }
93            else {
94                tokens.push(part);
95            }
96            prevPart = part;
97        });
98        return tokens;
99    }
100    join(tokens) {
101        // Tokens being joined here will always have appeared consecutively in the
102        // same text, so we can simply strip off the leading whitespace from all the
103        // tokens except the first (and except any whitespace-only tokens - but such
104        // a token will always be the first and only token anyway) and then join them
105        // and the whitespace around words and punctuation will end up correct.
106        return tokens.map((token, i) => {
107            if (i == 0) {
108                return token;
109            }
110            else {
111                return token.replace((/^\s+/), '');
112            }
113        }).join('');
114    }
115    postProcess(changes, options) {
116        if (!changes || options.oneChangePerToken) {
117            return changes;
118        }
119        let lastKeep = null;
120        // Change objects representing any insertion or deletion since the last
121        // "keep" change object. There can be at most one of each.
122        let insertion = null;
123        let deletion = null;
124        changes.forEach(change => {
125            if (change.added) {
126                insertion = change;
127            }
128            else if (change.removed) {
129                deletion = change;
130            }
131            else {
132                if (insertion || deletion) { // May be false at start of text
133                    dedupeWhitespaceInChangeObjects(lastKeep, deletion, insertion, change, options.intlSegmenter);
134                }
135                lastKeep = change;
136                insertion = null;
137                deletion = null;
138            }
139        });
140        if (insertion || deletion) {
141            dedupeWhitespaceInChangeObjects(lastKeep, deletion, insertion, null, options.intlSegmenter);
142        }
143        return changes;
144    }
145}
146export const wordDiff = new WordDiff();
147export function diffWords(oldStr, newStr, options) {
148    // This option has never been documented and never will be (it's clearer to
149    // just call `diffWordsWithSpace` directly if you need that behavior), but
150    // has existed in jsdiff for a long time, so we retain support for it here
151    // for the sake of backwards compatibility.
152    if ((options === null || options === void 0 ? void 0 : options.ignoreWhitespace) != null && !options.ignoreWhitespace) {
153        return diffWordsWithSpace(oldStr, newStr, options);
154    }
155    return wordDiff.diff(oldStr, newStr, options);
156}
157function dedupeWhitespaceInChangeObjects(startKeep, deletion, insertion, endKeep, segmenter) {
158    // Before returning, we tidy up the leading and trailing whitespace of the
159    // change objects to eliminate cases where trailing whitespace in one object
160    // is repeated as leading whitespace in the next.
161    // Below are examples of the outcomes we want here to explain the code.
162    // I=insert, K=keep, D=delete
163    // 1. diffing 'foo bar baz' vs 'foo baz'
164    //    Prior to cleanup, we have K:'foo ' D:' bar ' K:' baz'
165    //    After cleanup, we want:   K:'foo ' D:'bar ' K:'baz'
166    //
167    // 2. Diffing 'foo bar baz' vs 'foo qux baz'
168    //    Prior to cleanup, we have K:'foo ' D:' bar ' I:' qux ' K:' baz'
169    //    After cleanup, we want K:'foo ' D:'bar' I:'qux' K:' baz'
170    //
171    // 3. Diffing 'foo\nbar baz' vs 'foo baz'
172    //    Prior to cleanup, we have K:'foo ' D:'\nbar ' K:' baz'
173    //    After cleanup, we want K'foo' D:'\nbar' K:' baz'
174    //
175    // 4. Diffing 'foo baz' vs 'foo\nbar baz'
176    //    Prior to cleanup, we have K:'foo\n' I:'\nbar ' K:' baz'
177    //    After cleanup, we ideally want K'foo' I:'\nbar' K:' baz'
178    //    but don't actually manage this currently (the pre-cleanup change
179    //    objects don't contain enough information to make it possible).
180    //
181    // 5. Diffing 'foo   bar baz' vs 'foo  baz'
182    //    Prior to cleanup, we have K:'foo  ' D:'   bar ' K:'  baz'
183    //    After cleanup, we want K:'foo  ' D:' bar ' K:'baz'
184    //
185    // Our handling is unavoidably imperfect in the case where there's a single
186    // indel between keeps and the whitespace has changed. For instance, consider
187    // diffing 'foo\tbar\nbaz' vs 'foo baz'. Unless we create an extra change
188    // object to represent the insertion of the space character (which isn't even
189    // a token), we have no way to avoid losing information about the texts'
190    // original whitespace in the result we return. Still, we do our best to
191    // output something that will look sensible if we e.g. print it with
192    // insertions in green and deletions in red.
193    // Between two "keep" change objects (or before the first or after the last
194    // change object), we can have either:
195    // * A "delete" followed by an "insert"
196    // * Just an "insert"
197    // * Just a "delete"
198    // We handle the three cases separately.
199    if (deletion && insertion) {
200        const [oldWsPrefix, oldWsSuffix] = leadingAndTrailingWs(deletion.value, segmenter);
201        const [newWsPrefix, newWsSuffix] = leadingAndTrailingWs(insertion.value, segmenter);
202        if (startKeep) {
203            const commonWsPrefix = longestCommonPrefix(oldWsPrefix, newWsPrefix);
204            startKeep.value = replaceSuffix(startKeep.value, newWsPrefix, commonWsPrefix);
205            deletion.value = removePrefix(deletion.value, commonWsPrefix);
206            insertion.value = removePrefix(insertion.value, commonWsPrefix);
207        }
208        if (endKeep) {
209            const commonWsSuffix = longestCommonSuffix(oldWsSuffix, newWsSuffix);
210            endKeep.value = replacePrefix(endKeep.value, newWsSuffix, commonWsSuffix);
211            deletion.value = removeSuffix(deletion.value, commonWsSuffix);
212            insertion.value = removeSuffix(insertion.value, commonWsSuffix);
213        }
214    }
215    else if (insertion) {
216        // The whitespaces all reflect what was in the new text rather than
217        // the old, so we essentially have no information about whitespace
218        // insertion or deletion. We just want to dedupe the whitespace.
219        // We do that by having each change object keep its trailing
220        // whitespace and deleting duplicate leading whitespace where
221        // present.
222        if (startKeep) {
223            const ws = leadingWs(insertion.value, segmenter);
224            insertion.value = insertion.value.substring(ws.length);
225        }
226        if (endKeep) {
227            const ws = leadingWs(endKeep.value, segmenter);
228            endKeep.value = endKeep.value.substring(ws.length);
229        }
230        // otherwise we've got a deletion and no insertion
231    }
232    else if (startKeep && endKeep) {
233        const newWsFull = leadingWs(endKeep.value, segmenter), [delWsStart, delWsEnd] = leadingAndTrailingWs(deletion.value, segmenter);
234        // Any whitespace that comes straight after startKeep in both the old and
235        // new texts, assign to startKeep and remove from the deletion.
236        const newWsStart = longestCommonPrefix(newWsFull, delWsStart);
237        deletion.value = removePrefix(deletion.value, newWsStart);
238        // Any whitespace that comes straight before endKeep in both the old and
239        // new texts, and hasn't already been assigned to startKeep, assign to
240        // endKeep and remove from the deletion.
241        const newWsEnd = longestCommonSuffix(removePrefix(newWsFull, newWsStart), delWsEnd);
242        deletion.value = removeSuffix(deletion.value, newWsEnd);
243        endKeep.value = replacePrefix(endKeep.value, newWsFull, newWsEnd);
244        // If there's any whitespace from the new text that HASN'T already been
245        // assigned, assign it to the start:
246        startKeep.value = replaceSuffix(startKeep.value, newWsFull, newWsFull.slice(0, newWsFull.length - newWsEnd.length));
247    }
248    else if (endKeep) {
249        // We are at the start of the text. Preserve all the whitespace on
250        // endKeep, and just remove whitespace from the end of deletion to the
251        // extent that it overlaps with the start of endKeep.
252        const endKeepWsPrefix = leadingWs(endKeep.value, segmenter);
253        const deletionWsSuffix = trailingWs(deletion.value, segmenter);
254        const overlap = maximumOverlap(deletionWsSuffix, endKeepWsPrefix);
255        deletion.value = removeSuffix(deletion.value, overlap);
256    }
257    else if (startKeep) {
258        // We are at the END of the text. Preserve all the whitespace on
259        // startKeep, and just remove whitespace from the start of deletion to
260        // the extent that it overlaps with the end of startKeep.
261        const startKeepWsSuffix = trailingWs(startKeep.value, segmenter);
262        const deletionWsPrefix = leadingWs(deletion.value, segmenter);
263        const overlap = maximumOverlap(startKeepWsSuffix, deletionWsPrefix);
264        deletion.value = removePrefix(deletion.value, overlap);
265    }
266}
267class WordsWithSpaceDiff extends Diff {
268    tokenize(value) {
269        // Slightly different to the tokenizeIncludingWhitespace regex used above in
270        // that this one treats each individual newline as a distinct token, rather
271        // than merging them into other surrounding whitespace. This was requested
272        // in https://github.com/kpdecker/jsdiff/issues/180 &
273        //    https://github.com/kpdecker/jsdiff/issues/211
274        const regex = new RegExp(`(\\r?\\n)|[${extendedWordChars}]+|[^\\S\\n\\r]+|[^${extendedWordChars}]`, 'ug');
275        return value.match(regex) || [];
276    }
277}
278export const wordsWithSpaceDiff = new WordsWithSpaceDiff();
279export function diffWordsWithSpace(oldStr, newStr, options) {
280    return wordsWithSpaceDiff.diff(oldStr, newStr, options);
281}
282 
codekingpro/portable-devtools · Team Ai