codekingpro/portable-devtools
115k
1"use strict";
2var __extends = (this && this.__extends) || (function () {
3 var extendStatics = function (d, b) {
4 extendStatics = Object.setPrototypeOf ||
5 ({ __proto__: [] } instanceof Array && function (d, b) { d.__proto__ = b; }) ||
6 function (d, b) { for (var p in b) if (Object.prototype.hasOwnProperty.call(b, p)) d[p] = b[p]; };
7 return extendStatics(d, b);
8 };
9 return function (d, b) {
10 if (typeof b !== "function" && b !== null)
11 throw new TypeError("Class extends value " + String(b) + " is not a constructor or null");
12 extendStatics(d, b);
13 function __() { this.constructor = d; }
14 d.prototype = b === null ? Object.create(b) : (__.prototype = b.prototype, new __());
15 };
16})();
17Object.defineProperty(exports, "__esModule", { value: true });
18exports.wordsWithSpaceDiff = exports.wordDiff = void 0;
19exports.diffWords = diffWords;
20exports.diffWordsWithSpace = diffWordsWithSpace;
21var base_js_1 = require("./base.js");
22var string_js_1 = require("../util/string.js");
23// Based on https://en.wikipedia.org/wiki/Latin_script_in_Unicode
24//
25// Chars/ranges counted as "word" characters by this regex are as follows:
26//
27// + U+00AD Soft hyphen
28// + 00C0–00FF (letters with diacritics from the Latin-1 Supplement), except:
29// - U+00D7 × Multiplication sign
30// - U+00F7 ÷ Division sign
31// + Latin Extended-A, 0100–017F
32// + Latin Extended-B, 0180–024F
33// + IPA Extensions, 0250–02AF
34// + Spacing Modifier Letters, 02B0–02FF, except:
35// - U+02C7 ˇ ˇ Caron
36// - U+02D8 ˘ ˘ Breve
37// - U+02D9 ˙ ˙ Dot Above
38// - U+02DA ˚ ˚ Ring Above
39// - U+02DB ˛ ˛ Ogonek
40// - U+02DC ˜ ˜ Small Tilde
41// - U+02DD ˝ ˝ Double Acute Accent
42// + Latin Extended Additional, 1E00–1EFF
43var extendedWordChars = 'a-zA-Z0-9_\\u{AD}\\u{C0}-\\u{D6}\\u{D8}-\\u{F6}\\u{F8}-\\u{2C6}\\u{2C8}-\\u{2D7}\\u{2DE}-\\u{2FF}\\u{1E00}-\\u{1EFF}';
44// Each token is one of the following:
45// - A punctuation mark plus the surrounding whitespace
46// - A word plus the surrounding whitespace
47// - Pure whitespace (but only in the special case where the entire text
48// is just whitespace)
49//
50// We have to include surrounding whitespace in the tokens because the two
51// alternative approaches produce horribly broken results:
52// * If we just discard the whitespace, we can't fully reproduce the original
53// text from the sequence of tokens and any attempt to render the diff will
54// get the whitespace wrong.
55// * If we have separate tokens for whitespace, then in a typical text every
56// second token will be a single space character. But this often results in
57// the optimal diff between two texts being a perverse one that preserves
58// the spaces between words but deletes and reinserts actual common words.
59// See https://github.com/kpdecker/jsdiff/issues/160#issuecomment-1866099640
60// for an example.
61//
62// Keeping the surrounding whitespace of course has implications for .equals
63// and .join, not just .tokenize.
64// This regex does NOT fully implement the tokenization rules described above.
65// Instead, it gives runs of whitespace their own "token". The tokenize method
66// then handles stitching whitespace tokens onto adjacent word or punctuation
67// tokens.
68var tokenizeIncludingWhitespace = new RegExp("[".concat(extendedWordChars, "]+|\\s+|[^").concat(extendedWordChars, "]"), 'ug');
69var WordDiff = /** @class */ (function (_super) {
70 __extends(WordDiff, _super);
71 function WordDiff() {
72 return _super !== null && _super.apply(this, arguments) || this;
73 }
74 WordDiff.prototype.equals = function (left, right, options) {
75 if (options.ignoreCase) {
76 left = left.toLowerCase();
77 right = right.toLowerCase();
78 }
79 return left.trim() === right.trim();
80 };
81 WordDiff.prototype.tokenize = function (value, options) {
82 if (options === void 0) { options = {}; }
83 var parts;
84 if (options.intlSegmenter) {
85 var segmenter = options.intlSegmenter;
86 if (segmenter.resolvedOptions().granularity != 'word') {
87 throw new Error('The segmenter passed must have a granularity of "word"');
88 }
89 // We want `parts` to be an array whose elements alternate between being
90 // pure whitespace and being pure non-whitespace. This is ALMOST what the
91 // segments returned by a word-based Intl.Segmenter already look like,
92 // and therefore we can ALMOST get what we want by simply doing...
93 // parts = Array.from(segmenter.segment(value), segment => segment.segment);
94 // ... but not QUITE, because there's of one annoying special case: every
95 // newline character gets its own segment, instead of sharing a segment
96 // with other surrounding whitespace. We therefore need to manually merge
97 // consecutive segments of whitespace into a single part:
98 parts = [];
99 for (var _i = 0, _a = Array.from(segmenter.segment(value)); _i < _a.length; _i++) {
100 var segmentObj = _a[_i];
101 var segment = segmentObj.segment;
102 if (parts.length && (/\s/).test(parts[parts.length - 1]) && (/\s/).test(segment)) {
103 parts[parts.length - 1] += segment;
104 }
105 else {
106 parts.push(segment);
107 }
108 }
109 }
110 else {
111 parts = value.match(tokenizeIncludingWhitespace) || [];
112 }
113 var tokens = [];
114 var prevPart = null;
115 parts.forEach(function (part) {
116 if ((/\s/).test(part)) {
117 if (prevPart == null) {
118 tokens.push(part);
119 }
120 else {
121 tokens.push(tokens.pop() + part);
122 }
123 }
124 else if (prevPart != null && (/\s/).test(prevPart)) {
125 if (tokens[tokens.length - 1] == prevPart) {
126 tokens.push(tokens.pop() + part);
127 }
128 else {
129 tokens.push(prevPart + part);
130 }
131 }
132 else {
133 tokens.push(part);
134 }
135 prevPart = part;
136 });
137 return tokens;
138 };
139 WordDiff.prototype.join = function (tokens) {
140 // Tokens being joined here will always have appeared consecutively in the
141 // same text, so we can simply strip off the leading whitespace from all the
142 // tokens except the first (and except any whitespace-only tokens - but such
143 // a token will always be the first and only token anyway) and then join them
144 // and the whitespace around words and punctuation will end up correct.
145 return tokens.map(function (token, i) {
146 if (i == 0) {
147 return token;
148 }
149 else {
150 return token.replace((/^\s+/), '');
151 }
152 }).join('');
153 };
154 WordDiff.prototype.postProcess = function (changes, options) {
155 if (!changes || options.oneChangePerToken) {
156 return changes;
157 }
158 var lastKeep = null;
159 // Change objects representing any insertion or deletion since the last
160 // "keep" change object. There can be at most one of each.
161 var insertion = null;
162 var deletion = null;
163 changes.forEach(function (change) {
164 if (change.added) {
165 insertion = change;
166 }
167 else if (change.removed) {
168 deletion = change;
169 }
170 else {
171 if (insertion || deletion) { // May be false at start of text
172 dedupeWhitespaceInChangeObjects(lastKeep, deletion, insertion, change);
173 }
174 lastKeep = change;
175 insertion = null;
176 deletion = null;
177 }
178 });
179 if (insertion || deletion) {
180 dedupeWhitespaceInChangeObjects(lastKeep, deletion, insertion, null);
181 }
182 return changes;
183 };
184 return WordDiff;
185}(base_js_1.default));
186exports.wordDiff = new WordDiff();
187function diffWords(oldStr, newStr, options) {
188 // This option has never been documented and never will be (it's clearer to
189 // just call `diffWordsWithSpace` directly if you need that behavior), but
190 // has existed in jsdiff for a long time, so we retain support for it here
191 // for the sake of backwards compatibility.
192 if ((options === null || options === void 0 ? void 0 : options.ignoreWhitespace) != null && !options.ignoreWhitespace) {
193 return diffWordsWithSpace(oldStr, newStr, options);
194 }
195 return exports.wordDiff.diff(oldStr, newStr, options);
196}
197function dedupeWhitespaceInChangeObjects(startKeep, deletion, insertion, endKeep) {
198 // Before returning, we tidy up the leading and trailing whitespace of the
199 // change objects to eliminate cases where trailing whitespace in one object
200 // is repeated as leading whitespace in the next.
201 // Below are examples of the outcomes we want here to explain the code.
202 // I=insert, K=keep, D=delete
203 // 1. diffing 'foo bar baz' vs 'foo baz'
204 // Prior to cleanup, we have K:'foo ' D:' bar ' K:' baz'
205 // After cleanup, we want: K:'foo ' D:'bar ' K:'baz'
206 //
207 // 2. Diffing 'foo bar baz' vs 'foo qux baz'
208 // Prior to cleanup, we have K:'foo ' D:' bar ' I:' qux ' K:' baz'
209 // After cleanup, we want K:'foo ' D:'bar' I:'qux' K:' baz'
210 //
211 // 3. Diffing 'foo\nbar baz' vs 'foo baz'
212 // Prior to cleanup, we have K:'foo ' D:'\nbar ' K:' baz'
213 // After cleanup, we want K'foo' D:'\nbar' K:' baz'
214 //
215 // 4. Diffing 'foo baz' vs 'foo\nbar baz'
216 // Prior to cleanup, we have K:'foo\n' I:'\nbar ' K:' baz'
217 // After cleanup, we ideally want K'foo' I:'\nbar' K:' baz'
218 // but don't actually manage this currently (the pre-cleanup change
219 // objects don't contain enough information to make it possible).
220 //
221 // 5. Diffing 'foo bar baz' vs 'foo baz'
222 // Prior to cleanup, we have K:'foo ' D:' bar ' K:' baz'
223 // After cleanup, we want K:'foo ' D:' bar ' K:'baz'
224 //
225 // Our handling is unavoidably imperfect in the case where there's a single
226 // indel between keeps and the whitespace has changed. For instance, consider
227 // diffing 'foo\tbar\nbaz' vs 'foo baz'. Unless we create an extra change
228 // object to represent the insertion of the space character (which isn't even
229 // a token), we have no way to avoid losing information about the texts'
230 // original whitespace in the result we return. Still, we do our best to
231 // output something that will look sensible if we e.g. print it with
232 // insertions in green and deletions in red.
233 // Between two "keep" change objects (or before the first or after the last
234 // change object), we can have either:
235 // * A "delete" followed by an "insert"
236 // * Just an "insert"
237 // * Just a "delete"
238 // We handle the three cases separately.
239 if (deletion && insertion) {
240 var oldWsPrefix = (0, string_js_1.leadingWs)(deletion.value);
241 var oldWsSuffix = (0, string_js_1.trailingWs)(deletion.value);
242 var newWsPrefix = (0, string_js_1.leadingWs)(insertion.value);
243 var newWsSuffix = (0, string_js_1.trailingWs)(insertion.value);
244 if (startKeep) {
245 var commonWsPrefix = (0, string_js_1.longestCommonPrefix)(oldWsPrefix, newWsPrefix);
246 startKeep.value = (0, string_js_1.replaceSuffix)(startKeep.value, newWsPrefix, commonWsPrefix);
247 deletion.value = (0, string_js_1.removePrefix)(deletion.value, commonWsPrefix);
248 insertion.value = (0, string_js_1.removePrefix)(insertion.value, commonWsPrefix);
249 }
250 if (endKeep) {
251 var commonWsSuffix = (0, string_js_1.longestCommonSuffix)(oldWsSuffix, newWsSuffix);
252 endKeep.value = (0, string_js_1.replacePrefix)(endKeep.value, newWsSuffix, commonWsSuffix);
253 deletion.value = (0, string_js_1.removeSuffix)(deletion.value, commonWsSuffix);
254 insertion.value = (0, string_js_1.removeSuffix)(insertion.value, commonWsSuffix);
255 }
256 }
257 else if (insertion) {
258 // The whitespaces all reflect what was in the new text rather than
259 // the old, so we essentially have no information about whitespace
260 // insertion or deletion. We just want to dedupe the whitespace.
261 // We do that by having each change object keep its trailing
262 // whitespace and deleting duplicate leading whitespace where
263 // present.
264 if (startKeep) {
265 var ws = (0, string_js_1.leadingWs)(insertion.value);
266 insertion.value = insertion.value.substring(ws.length);
267 }
268 if (endKeep) {
269 var ws = (0, string_js_1.leadingWs)(endKeep.value);
270 endKeep.value = endKeep.value.substring(ws.length);
271 }
272 // otherwise we've got a deletion and no insertion
273 }
274 else if (startKeep && endKeep) {
275 var newWsFull = (0, string_js_1.leadingWs)(endKeep.value), delWsStart = (0, string_js_1.leadingWs)(deletion.value), delWsEnd = (0, string_js_1.trailingWs)(deletion.value);
276 // Any whitespace that comes straight after startKeep in both the old and
277 // new texts, assign to startKeep and remove from the deletion.
278 var newWsStart = (0, string_js_1.longestCommonPrefix)(newWsFull, delWsStart);
279 deletion.value = (0, string_js_1.removePrefix)(deletion.value, newWsStart);
280 // Any whitespace that comes straight before endKeep in both the old and
281 // new texts, and hasn't already been assigned to startKeep, assign to
282 // endKeep and remove from the deletion.
283 var newWsEnd = (0, string_js_1.longestCommonSuffix)((0, string_js_1.removePrefix)(newWsFull, newWsStart), delWsEnd);
284 deletion.value = (0, string_js_1.removeSuffix)(deletion.value, newWsEnd);
285 endKeep.value = (0, string_js_1.replacePrefix)(endKeep.value, newWsFull, newWsEnd);
286 // If there's any whitespace from the new text that HASN'T already been
287 // assigned, assign it to the start:
288 startKeep.value = (0, string_js_1.replaceSuffix)(startKeep.value, newWsFull, newWsFull.slice(0, newWsFull.length - newWsEnd.length));
289 }
290 else if (endKeep) {
291 // We are at the start of the text. Preserve all the whitespace on
292 // endKeep, and just remove whitespace from the end of deletion to the
293 // extent that it overlaps with the start of endKeep.
294 var endKeepWsPrefix = (0, string_js_1.leadingWs)(endKeep.value);
295 var deletionWsSuffix = (0, string_js_1.trailingWs)(deletion.value);
296 var overlap = (0, string_js_1.maximumOverlap)(deletionWsSuffix, endKeepWsPrefix);
297 deletion.value = (0, string_js_1.removeSuffix)(deletion.value, overlap);
298 }
299 else if (startKeep) {
300 // We are at the END of the text. Preserve all the whitespace on
301 // startKeep, and just remove whitespace from the start of deletion to
302 // the extent that it overlaps with the end of startKeep.
303 var startKeepWsSuffix = (0, string_js_1.trailingWs)(startKeep.value);
304 var deletionWsPrefix = (0, string_js_1.leadingWs)(deletion.value);
305 var overlap = (0, string_js_1.maximumOverlap)(startKeepWsSuffix, deletionWsPrefix);
306 deletion.value = (0, string_js_1.removePrefix)(deletion.value, overlap);
307 }
308}
309var WordsWithSpaceDiff = /** @class */ (function (_super) {
310 __extends(WordsWithSpaceDiff, _super);
311 function WordsWithSpaceDiff() {
312 return _super !== null && _super.apply(this, arguments) || this;
313 }
314 WordsWithSpaceDiff.prototype.tokenize = function (value) {
315 // Slightly different to the tokenizeIncludingWhitespace regex used above in
316 // that this one treats each individual newline as a distinct token, rather
317 // than merging them into other surrounding whitespace. This was requested
318 // in https://github.com/kpdecker/jsdiff/issues/180 &
319 // https://github.com/kpdecker/jsdiff/issues/211
320 var regex = new RegExp("(\\r?\\n)|[".concat(extendedWordChars, "]+|[^\\S\\n\\r]+|[^").concat(extendedWordChars, "]"), 'ug');
321 return value.match(regex) || [];
322 };
323 return WordsWithSpaceDiff;
324}(base_js_1.default));
325exports.wordsWithSpaceDiff = new WordsWithSpaceDiff();
326function diffWordsWithSpace(oldStr, newStr, options) {
327 return exports.wordsWithSpaceDiff.diff(oldStr, newStr, options);
328}
329 