Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes15kdownloads
sentence.js68 linesDownload Raw Back to diff
1"use strict";
2var __extends = (this && this.__extends) || (function () {
3    var extendStatics = function (d, b) {
4        extendStatics = Object.setPrototypeOf ||
5            ({ __proto__: [] } instanceof Array && function (d, b) { d.__proto__ = b; }) ||
6            function (d, b) { for (var p in b) if (Object.prototype.hasOwnProperty.call(b, p)) d[p] = b[p]; };
7        return extendStatics(d, b);
8    };
9    return function (d, b) {
10        if (typeof b !== "function" && b !== null)
11            throw new TypeError("Class extends value " + String(b) + " is not a constructor or null");
12        extendStatics(d, b);
13        function __() { this.constructor = d; }
14        d.prototype = b === null ? Object.create(b) : (__.prototype = b.prototype, new __());
15    };
16})();
17Object.defineProperty(exports, "__esModule", { value: true });
18exports.sentenceDiff = void 0;
19exports.diffSentences = diffSentences;
20var base_js_1 = require("./base.js");
21function isSentenceEndPunct(char) {
22    return char == '.' || char == '!' || char == '?';
23}
24var SentenceDiff = /** @class */ (function (_super) {
25    __extends(SentenceDiff, _super);
26    function SentenceDiff() {
27        return _super !== null && _super.apply(this, arguments) || this;
28    }
29    SentenceDiff.prototype.tokenize = function (value) {
30        var _a;
31        // If in future we drop support for environments that don't support lookbehinds, we can replace
32        // this entire function with:
33        //     return value.split(/(?<=[.!?])(\s+|$)/);
34        // but until then, for similar reasons to the trailingWs function in string.ts, we are forced
35        // to do this verbosely "by hand" instead of using a regex.
36        var result = [];
37        var tokenStartI = 0;
38        for (var i = 0; i < value.length; i++) {
39            if (i == value.length - 1) {
40                result.push(value.slice(tokenStartI));
41                break;
42            }
43            if (isSentenceEndPunct(value[i]) && value[i + 1].match(/\s/)) {
44                // We've hit a sentence break - i.e. a punctuation mark followed by whitespace.
45                // We now want to push TWO tokens to the result:
46                // 1. the sentence
47                result.push(value.slice(tokenStartI, i + 1));
48                // 2. the whitespace
49                i = tokenStartI = i + 1;
50                while ((_a = value[i + 1]) === null || _a === void 0 ? void 0 : _a.match(/\s/)) {
51                    i++;
52                }
53                result.push(value.slice(tokenStartI, i + 1));
54                // Then the next token (a sentence) starts on the character after the whitespace.
55                // (It's okay if this is off the end of the string - then the outer loop will terminate
56                // here anyway.)
57                tokenStartI = i + 1;
58            }
59        }
60        return result;
61    };
62    return SentenceDiff;
63}(base_js_1.default));
64exports.sentenceDiff = new SentenceDiff();
65function diffSentences(oldStr, newStr, options) {
66    return exports.sentenceDiff.diff(oldStr, newStr, options);
67}
68 
codekingpro/portable-devtools · Team Ai