Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes15kdownloads
word.js313 linesDownload Raw Back to diff
1"use strict";
2var __extends = (this && this.__extends) || (function () {
3    var extendStatics = function (d, b) {
4        extendStatics = Object.setPrototypeOf ||
5            ({ __proto__: [] } instanceof Array && function (d, b) { d.__proto__ = b; }) ||
6            function (d, b) { for (var p in b) if (Object.prototype.hasOwnProperty.call(b, p)) d[p] = b[p]; };
7        return extendStatics(d, b);
8    };
9    return function (d, b) {
10        if (typeof b !== "function" && b !== null)
11            throw new TypeError("Class extends value " + String(b) + " is not a constructor or null");
12        extendStatics(d, b);
13        function __() { this.constructor = d; }
14        d.prototype = b === null ? Object.create(b) : (__.prototype = b.prototype, new __());
15    };
16})();
17Object.defineProperty(exports, "__esModule", { value: true });
18exports.wordsWithSpaceDiff = exports.wordDiff = void 0;
19exports.diffWords = diffWords;
20exports.diffWordsWithSpace = diffWordsWithSpace;
21var base_js_1 = require("./base.js");
22var string_js_1 = require("../util/string.js");
23// Based on https://en.wikipedia.org/wiki/Latin_script_in_Unicode
24//
25// Chars/ranges counted as "word" characters by this regex are as follows:
26//
27// + U+00AD  Soft hyphen
28// + 00C0–00FF (letters with diacritics from the Latin-1 Supplement), except:
29//   - U+00D7  × Multiplication sign
30//   - U+00F7  ÷ Division sign
31// + Latin Extended-A, 0100–017F
32// + Latin Extended-B, 0180–024F
33// + IPA Extensions, 0250–02AF
34// + Spacing Modifier Letters, 02B0–02FF, except:
35//   - U+02C7  ˇ ˇ  Caron
36//   - U+02D8  ˘ ˘  Breve
37//   - U+02D9  ˙ ˙  Dot Above
38//   - U+02DA  ˚ ˚  Ring Above
39//   - U+02DB  ˛ ˛  Ogonek
40//   - U+02DC  ˜ ˜  Small Tilde
41//   - U+02DD  ˝ ˝  Double Acute Accent
42// + Latin Extended Additional, 1E00–1EFF
43var extendedWordChars = 'a-zA-Z0-9_\\u{AD}\\u{C0}-\\u{D6}\\u{D8}-\\u{F6}\\u{F8}-\\u{2C6}\\u{2C8}-\\u{2D7}\\u{2DE}-\\u{2FF}\\u{1E00}-\\u{1EFF}';
44// Each token is one of the following:
45// - A punctuation mark plus the surrounding whitespace
46// - A word plus the surrounding whitespace
47// - Pure whitespace (but only in the special case where the entire text
48//   is just whitespace)
49//
50// We have to include surrounding whitespace in the tokens because the two
51// alternative approaches produce horribly broken results:
52// * If we just discard the whitespace, we can't fully reproduce the original
53//   text from the sequence of tokens and any attempt to render the diff will
54//   get the whitespace wrong.
55// * If we have separate tokens for whitespace, then in a typical text every
56//   second token will be a single space character. But this often results in
57//   the optimal diff between two texts being a perverse one that preserves
58//   the spaces between words but deletes and reinserts actual common words.
59//   See https://github.com/kpdecker/jsdiff/issues/160#issuecomment-1866099640
60//   for an example.
61//
62// Keeping the surrounding whitespace of course has implications for .equals
63// and .join, not just .tokenize.
64// This regex does NOT fully implement the tokenization rules described above.
65// Instead, it gives runs of whitespace their own "token". The tokenize method
66// then handles stitching whitespace tokens onto adjacent word or punctuation
67// tokens.
68var tokenizeIncludingWhitespace = new RegExp("[".concat(extendedWordChars, "]+|\\s+|[^").concat(extendedWordChars, "]"), 'ug');
69var WordDiff = /** @class */ (function (_super) {
70    __extends(WordDiff, _super);
71    function WordDiff() {
72        return _super !== null && _super.apply(this, arguments) || this;
73    }
74    WordDiff.prototype.equals = function (left, right, options) {
75        if (options.ignoreCase) {
76            left = left.toLowerCase();
77            right = right.toLowerCase();
78        }
79        return left.trim() === right.trim();
80    };
81    WordDiff.prototype.tokenize = function (value, options) {
82        if (options === void 0) { options = {}; }
83        var parts;
84        if (options.intlSegmenter) {
85            var segmenter = options.intlSegmenter;
86            if (segmenter.resolvedOptions().granularity != 'word') {
87                throw new Error('The segmenter passed must have a granularity of "word"');
88            }
89            // We want `parts` to be an array whose elements alternate between being
90            // pure whitespace and being pure non-whitespace. This is ALMOST what the
91            // segments returned by a word-based Intl.Segmenter already look like,
92            // but not quite - see explanation in the docs of our custom segment()
93            // function.
94            parts = (0, string_js_1.segment)(value, segmenter);
95        }
96        else {
97            parts = value.match(tokenizeIncludingWhitespace) || [];
98        }
99        var tokens = [];
100        var prevPart = null;
101        parts.forEach(function (part) {
102            if ((/\s/).test(part)) {
103                if (prevPart == null) {
104                    tokens.push(part);
105                }
106                else {
107                    tokens.push(tokens.pop() + part);
108                }
109            }
110            else if (prevPart != null && (/\s/).test(prevPart)) {
111                if (tokens[tokens.length - 1] == prevPart) {
112                    tokens.push(tokens.pop() + part);
113                }
114                else {
115                    tokens.push(prevPart + part);
116                }
117            }
118            else {
119                tokens.push(part);
120            }
121            prevPart = part;
122        });
123        return tokens;
124    };
125    WordDiff.prototype.join = function (tokens) {
126        // Tokens being joined here will always have appeared consecutively in the
127        // same text, so we can simply strip off the leading whitespace from all the
128        // tokens except the first (and except any whitespace-only tokens - but such
129        // a token will always be the first and only token anyway) and then join them
130        // and the whitespace around words and punctuation will end up correct.
131        return tokens.map(function (token, i) {
132            if (i == 0) {
133                return token;
134            }
135            else {
136                return token.replace((/^\s+/), '');
137            }
138        }).join('');
139    };
140    WordDiff.prototype.postProcess = function (changes, options) {
141        if (!changes || options.oneChangePerToken) {
142            return changes;
143        }
144        var lastKeep = null;
145        // Change objects representing any insertion or deletion since the last
146        // "keep" change object. There can be at most one of each.
147        var insertion = null;
148        var deletion = null;
149        changes.forEach(function (change) {
150            if (change.added) {
151                insertion = change;
152            }
153            else if (change.removed) {
154                deletion = change;
155            }
156            else {
157                if (insertion || deletion) { // May be false at start of text
158                    dedupeWhitespaceInChangeObjects(lastKeep, deletion, insertion, change, options.intlSegmenter);
159                }
160                lastKeep = change;
161                insertion = null;
162                deletion = null;
163            }
164        });
165        if (insertion || deletion) {
166            dedupeWhitespaceInChangeObjects(lastKeep, deletion, insertion, null, options.intlSegmenter);
167        }
168        return changes;
169    };
170    return WordDiff;
171}(base_js_1.default));
172exports.wordDiff = new WordDiff();
173function diffWords(oldStr, newStr, options) {
174    // This option has never been documented and never will be (it's clearer to
175    // just call `diffWordsWithSpace` directly if you need that behavior), but
176    // has existed in jsdiff for a long time, so we retain support for it here
177    // for the sake of backwards compatibility.
178    if ((options === null || options === void 0 ? void 0 : options.ignoreWhitespace) != null && !options.ignoreWhitespace) {
179        return diffWordsWithSpace(oldStr, newStr, options);
180    }
181    return exports.wordDiff.diff(oldStr, newStr, options);
182}
183function dedupeWhitespaceInChangeObjects(startKeep, deletion, insertion, endKeep, segmenter) {
184    // Before returning, we tidy up the leading and trailing whitespace of the
185    // change objects to eliminate cases where trailing whitespace in one object
186    // is repeated as leading whitespace in the next.
187    // Below are examples of the outcomes we want here to explain the code.
188    // I=insert, K=keep, D=delete
189    // 1. diffing 'foo bar baz' vs 'foo baz'
190    //    Prior to cleanup, we have K:'foo ' D:' bar ' K:' baz'
191    //    After cleanup, we want:   K:'foo ' D:'bar ' K:'baz'
192    //
193    // 2. Diffing 'foo bar baz' vs 'foo qux baz'
194    //    Prior to cleanup, we have K:'foo ' D:' bar ' I:' qux ' K:' baz'
195    //    After cleanup, we want K:'foo ' D:'bar' I:'qux' K:' baz'
196    //
197    // 3. Diffing 'foo\nbar baz' vs 'foo baz'
198    //    Prior to cleanup, we have K:'foo ' D:'\nbar ' K:' baz'
199    //    After cleanup, we want K'foo' D:'\nbar' K:' baz'
200    //
201    // 4. Diffing 'foo baz' vs 'foo\nbar baz'
202    //    Prior to cleanup, we have K:'foo\n' I:'\nbar ' K:' baz'
203    //    After cleanup, we ideally want K'foo' I:'\nbar' K:' baz'
204    //    but don't actually manage this currently (the pre-cleanup change
205    //    objects don't contain enough information to make it possible).
206    //
207    // 5. Diffing 'foo   bar baz' vs 'foo  baz'
208    //    Prior to cleanup, we have K:'foo  ' D:'   bar ' K:'  baz'
209    //    After cleanup, we want K:'foo  ' D:' bar ' K:'baz'
210    //
211    // Our handling is unavoidably imperfect in the case where there's a single
212    // indel between keeps and the whitespace has changed. For instance, consider
213    // diffing 'foo\tbar\nbaz' vs 'foo baz'. Unless we create an extra change
214    // object to represent the insertion of the space character (which isn't even
215    // a token), we have no way to avoid losing information about the texts'
216    // original whitespace in the result we return. Still, we do our best to
217    // output something that will look sensible if we e.g. print it with
218    // insertions in green and deletions in red.
219    // Between two "keep" change objects (or before the first or after the last
220    // change object), we can have either:
221    // * A "delete" followed by an "insert"
222    // * Just an "insert"
223    // * Just a "delete"
224    // We handle the three cases separately.
225    if (deletion && insertion) {
226        var _a = (0, string_js_1.leadingAndTrailingWs)(deletion.value, segmenter), oldWsPrefix = _a[0], oldWsSuffix = _a[1];
227        var _b = (0, string_js_1.leadingAndTrailingWs)(insertion.value, segmenter), newWsPrefix = _b[0], newWsSuffix = _b[1];
228        if (startKeep) {
229            var commonWsPrefix = (0, string_js_1.longestCommonPrefix)(oldWsPrefix, newWsPrefix);
230            startKeep.value = (0, string_js_1.replaceSuffix)(startKeep.value, newWsPrefix, commonWsPrefix);
231            deletion.value = (0, string_js_1.removePrefix)(deletion.value, commonWsPrefix);
232            insertion.value = (0, string_js_1.removePrefix)(insertion.value, commonWsPrefix);
233        }
234        if (endKeep) {
235            var commonWsSuffix = (0, string_js_1.longestCommonSuffix)(oldWsSuffix, newWsSuffix);
236            endKeep.value = (0, string_js_1.replacePrefix)(endKeep.value, newWsSuffix, commonWsSuffix);
237            deletion.value = (0, string_js_1.removeSuffix)(deletion.value, commonWsSuffix);
238            insertion.value = (0, string_js_1.removeSuffix)(insertion.value, commonWsSuffix);
239        }
240    }
241    else if (insertion) {
242        // The whitespaces all reflect what was in the new text rather than
243        // the old, so we essentially have no information about whitespace
244        // insertion or deletion. We just want to dedupe the whitespace.
245        // We do that by having each change object keep its trailing
246        // whitespace and deleting duplicate leading whitespace where
247        // present.
248        if (startKeep) {
249            var ws = (0, string_js_1.leadingWs)(insertion.value, segmenter);
250            insertion.value = insertion.value.substring(ws.length);
251        }
252        if (endKeep) {
253            var ws = (0, string_js_1.leadingWs)(endKeep.value, segmenter);
254            endKeep.value = endKeep.value.substring(ws.length);
255        }
256        // otherwise we've got a deletion and no insertion
257    }
258    else if (startKeep && endKeep) {
259        var newWsFull = (0, string_js_1.leadingWs)(endKeep.value, segmenter), _c = (0, string_js_1.leadingAndTrailingWs)(deletion.value, segmenter), delWsStart = _c[0], delWsEnd = _c[1];
260        // Any whitespace that comes straight after startKeep in both the old and
261        // new texts, assign to startKeep and remove from the deletion.
262        var newWsStart = (0, string_js_1.longestCommonPrefix)(newWsFull, delWsStart);
263        deletion.value = (0, string_js_1.removePrefix)(deletion.value, newWsStart);
264        // Any whitespace that comes straight before endKeep in both the old and
265        // new texts, and hasn't already been assigned to startKeep, assign to
266        // endKeep and remove from the deletion.
267        var newWsEnd = (0, string_js_1.longestCommonSuffix)((0, string_js_1.removePrefix)(newWsFull, newWsStart), delWsEnd);
268        deletion.value = (0, string_js_1.removeSuffix)(deletion.value, newWsEnd);
269        endKeep.value = (0, string_js_1.replacePrefix)(endKeep.value, newWsFull, newWsEnd);
270        // If there's any whitespace from the new text that HASN'T already been
271        // assigned, assign it to the start:
272        startKeep.value = (0, string_js_1.replaceSuffix)(startKeep.value, newWsFull, newWsFull.slice(0, newWsFull.length - newWsEnd.length));
273    }
274    else if (endKeep) {
275        // We are at the start of the text. Preserve all the whitespace on
276        // endKeep, and just remove whitespace from the end of deletion to the
277        // extent that it overlaps with the start of endKeep.
278        var endKeepWsPrefix = (0, string_js_1.leadingWs)(endKeep.value, segmenter);
279        var deletionWsSuffix = (0, string_js_1.trailingWs)(deletion.value, segmenter);
280        var overlap = (0, string_js_1.maximumOverlap)(deletionWsSuffix, endKeepWsPrefix);
281        deletion.value = (0, string_js_1.removeSuffix)(deletion.value, overlap);
282    }
283    else if (startKeep) {
284        // We are at the END of the text. Preserve all the whitespace on
285        // startKeep, and just remove whitespace from the start of deletion to
286        // the extent that it overlaps with the end of startKeep.
287        var startKeepWsSuffix = (0, string_js_1.trailingWs)(startKeep.value, segmenter);
288        var deletionWsPrefix = (0, string_js_1.leadingWs)(deletion.value, segmenter);
289        var overlap = (0, string_js_1.maximumOverlap)(startKeepWsSuffix, deletionWsPrefix);
290        deletion.value = (0, string_js_1.removePrefix)(deletion.value, overlap);
291    }
292}
293var WordsWithSpaceDiff = /** @class */ (function (_super) {
294    __extends(WordsWithSpaceDiff, _super);
295    function WordsWithSpaceDiff() {
296        return _super !== null && _super.apply(this, arguments) || this;
297    }
298    WordsWithSpaceDiff.prototype.tokenize = function (value) {
299        // Slightly different to the tokenizeIncludingWhitespace regex used above in
300        // that this one treats each individual newline as a distinct token, rather
301        // than merging them into other surrounding whitespace. This was requested
302        // in https://github.com/kpdecker/jsdiff/issues/180 &
303        //    https://github.com/kpdecker/jsdiff/issues/211
304        var regex = new RegExp("(\\r?\\n)|[".concat(extendedWordChars, "]+|[^\\S\\n\\r]+|[^").concat(extendedWordChars, "]"), 'ug');
305        return value.match(regex) || [];
306    };
307    return WordsWithSpaceDiff;
308}(base_js_1.default));
309exports.wordsWithSpaceDiff = new WordsWithSpaceDiff();
310function diffWordsWithSpace(oldStr, newStr, options) {
311    return exports.wordsWithSpaceDiff.diff(oldStr, newStr, options);
312}
313 
codekingpro/portable-devtools · Team Ai