Kashtan/Detect_Edits_in_AI-Generated_Text
0
1import logging
2import spacy
3import re
4import numpy as np
5from src.SentenceParser import SentenceParser
6
7class PrepareSentenceContext(object):
8 """
9 Parse text and extract length and context information
10
11 This information is needed for evaluating log-perplexity of the text with respect to a language model
12 and later on to test the likelihood that the sentence was sampled from the model with the relevant context.
13 """
14
15 def __init__(self, sentence_parser='spacy', context_policy=None, context=None):
16 if sentence_parser == 'spacy':
17 self.nlp = spacy.load("en_core_web_sm", disable=["tagger", "attribute_ruler", "lemmatizer", "ner"])
18 if sentence_parser == 'regex':
19 logging.warning("Regex-based parser is not good at breaking sentences like 'Dr. Stone', etc.")
20 self.nlp = SentenceParser()
21
22 self.sentence_parser_name = sentence_parser
23
24 self.context_policy = context_policy
25 self.context = context
26
27 def __call__(self, text):
28 return self.parse_sentences(text)
29
30 def parse_sentences(self, text):
31 pattern_close = r"(.*?)</edit>"
32 pattern_open = r"<edit>(.*?)"
33 MIN_TOKEN_LEN = 3
34
35 texts = []
36 tags = []
37 lengths = []
38 contexts = []
39
40 def update_sent(sent_text, tag, sent_length):
41 texts.append(sent_text)
42 tags.append(tag)
43 lengths.append(sent_length)
44 if self.context is not None:
45 context = self.context
46 elif self.context_policy is None:
47 context = None
48 elif self.context_policy == 'previous_sentence' and len(texts) > 0:
49 context = texts[-1]
50 else:
51 context = None
52 contexts.append(context)
53
54 curr_tag = None
55 parsed = self.nlp(text)
56 for s in parsed.sents:
57 prev_tag = curr_tag
58 matches_close = re.findall(pattern_close, s.text)
59 matches_open = re.findall(pattern_open, s.text)
60 matches_between = re.findall(r"<edit>(.*?)</edit>", s.text)
61
62 logging.debug(f"Current sentence: {s.text}")
63 logging.debug(f"Matches open: {matches_open}")
64 logging.debug(f"Matches close: {matches_close}")
65 logging.debug(f"Matches between: {matches_between}")
66 if len(matches_close)>0 and len(matches_open)>0:
67 logging.debug("Found an opening and a closing tag in the same sentence.")
68 if prev_tag is None and len(matches_open[0]) >= MIN_TOKEN_LEN:
69 logging.debug("Openning followed by closing with some text in between.")
70 update_sent(matches_open[0], "<edit>", len(s)-2)
71 curr_tag = None
72 if prev_tag == "<edit>" and len(matches_close[0]) >= MIN_TOKEN_LEN:
73 logging.warning(f"Wierd case: closing/openning followed by openning in sentence {len(texts)}")
74 update_sent(matches_close[0], prev_tag, len(s)-1)
75 curr_tag = None
76 if prev_tag == "</edit>":
77 logging.debug("Closing followed by openning.")
78 curr_tag = "<edit>"
79 if len(matches_between[0]) > MIN_TOKEN_LEN:
80 update_sent(matches_between[0], None, len(s)-2)
81 elif len(matches_open) > 0:
82 curr_tag = "<edit>"
83 assert prev_tag is None, f"Found an opening tag without a closing tag in sentence num. {len(texts)}"
84 if len(matches_open[0]) >= MIN_TOKEN_LEN:
85 # text and tag are in the same sentence
86 sent_text = matches_open[0]
87 update_sent(sent_text, curr_tag, len(s)-1)
88 elif len(matches_close) > 0:
89 curr_tag = "</edit>"
90 assert prev_tag == "<edit>", f"Found a closing tag without an opening tag in sentence num. {len(texts)}"
91 if len(matches_close[0]) >= MIN_TOKEN_LEN:
92 # text and tag are in the same sentence
93 update_sent(matches_close[0], prev_tag, len(s)-1)
94 curr_tag = None
95 else:
96 #if len(matches_close)==0 and len(matches_open)==0:
97 # no tag
98 update_sent(s.text, curr_tag, len(s))
99 return {'text': texts, 'length': lengths, 'context': contexts, 'tag': tags,
100 'number_in_par': np.arange(1,1+len(texts))}
101
102 def REMOVE_parse_sentences(self, text):
103 texts = []
104 contexts = []
105 lengths = []
106 tags = []
107 num_in_par = []
108 previous = None
109
110 text = re.sub("(</?[a-zA-Z0-9 ]+>\.?)\s+", r"\1.\n", text) # to make sure that tags are in separate sentences
111 #text = re.sub("(</[a-zA-Z0-9 ]+>\.?)\s+", r"\n\1.\n", text) # to make sure that tags are in separate sentences
112
113 parsed = self.nlp(text)
114
115 running_sent_num = 0
116 curr_tag = None
117 for i, sent in enumerate(parsed.sents):
118 # Here we try to track HTML-like tags. There might be
119 # some issues because spacy sentence parser has unexpected behavior when it comes to newlines
120 all_tags = re.findall(r"(</?[a-zA-Z0-9 ]+>)", str(sent))
121 if len(all_tags) > 1:
122 logging.error(f"More than one tag in sentence {i}: {all_tags}")
123 exit(1)
124 if len(all_tags) == 1:
125 tag = all_tags[0]
126 if tag[:2] == '</': # a closing tag
127 if curr_tag is None:
128 logging.warning(f"Closing tag without an opening tag in sentence {i}: {sent}")
129 else:
130 curr_tag = None
131 else:
132 if curr_tag is not None:
133 logging.warning(f"Opening tag without a closing tag in sentence {i}: {sent}")
134 else:
135 curr_tag = tag
136 else: # if text is not a tag
137 sent_text = str(sent)
138 sent_length = len(sent)
139
140 texts.append(sent_text)
141 running_sent_num += 1
142 num_in_par.append(running_sent_num)
143 tags.append(curr_tag)
144 lengths.append(sent_length)
145
146 if self.context is not None:
147 context = self.context
148 elif self.context_policy is None:
149 context = None
150 elif self.context_policy == 'previous_sentence':
151 context = previous
152 previous = sent_text
153 else:
154 context = None
155
156 contexts.append(context)
157 return {'text': texts, 'length': lengths, 'context': contexts, 'tag': tags,
158 'number_in_par': num_in_par}