chris32/Text-Intelligence-Real-State
1
1# Datetime2import datetime3# Manipulate4import re5import json6import pandas as pd7# App8import gradio as gr9# GLiNER Model10from gliner import GLiNER11# Transformers12from transformers import pipeline13 14# Load GLiNER Model15model = GLiNER.from_pretrained("chris32/gliner_multi_pii_real_state-v2")16model.eval()17 18# BERT Model19model_name = "chris32/distilbert-base-spanish-uncased-finetuned-text-intelligence"20pipe = pipeline(model = model_name, device = "cpu")21 22# Global Variables: For Post Cleaning Inferences23YEAR_OF_REMODELING_LIMIT = 10024CURRENT_YEAR = int(datetime.date.today().year)25SCORE_LIMIT_SIMILARITY_NAMES = 7026 27def clean_text(text):28 # Replace HTML line breaks with the specified character29 replacement_char = " # "30 text = re.sub(r'<br\s*\/?>', replacement_char, text)31 32 # Remove HTML tags and special characters33 cleaned_text = re.sub(r'<[^>]*>', '', text)34 cleaned_text = re.sub(r' ', ' ', cleaned_text)35 cleaned_text = re.sub(r'&', '&', cleaned_text)36 37 # Drop punctuation marks38 #regex = '[\\!\\"\\#\\$\\%\\&\\\'\\(\\)\\*\\+\\,\\-\\.\\/\\:\\;\\<\\=\\>\\?\\@\\[\\\\\\]\\^_\\`\\{\\|\\}\\~]'39 #cleaned_text = re.sub(regex , ' ', cleaned_text)40 41 # Replace multiple spaces with a single one42 cleaned_text = re.sub(r'\s+', ' ', cleaned_text)43 44 # Remove leading and trailing spaces45 cleaned_text = cleaned_text.strip()46 47 # Replace Duplicated "." and ","48 cleaned_text = cleaned_text.replace("..", ".").replace(",,", ",")49 50 return cleaned_text51 52def format_gliner_predictions(prediction):53 if len(prediction) > 0: 54 # Select the Entity value with the Greater Score for each Entity Name55 prediction_df = pd.DataFrame(prediction)\56 .sort_values("score", ascending = False)\57 .drop_duplicates(subset = "label", keep = "first")58 59 # Add Position Column60 prediction_df["position"] = prediction_df.apply(lambda x: (x["start"], x["end"]) ,axis = 1)61 62 # Add Columns Label for Text and Probability63 prediction_df["label_text"] = prediction_df["label"].apply(lambda x: f"pred_{x}")64 prediction_df["label_prob"] = prediction_df["label"].apply(lambda x: f"prob_{x}")65 prediction_df["label_position"] = prediction_df["label"].apply(lambda x: f"pos_{x}")66 67 # Format Predictions68 entities = prediction_df.set_index("label_text")["text"].to_dict()69 entities_probs = prediction_df.set_index("label_prob")["score"].to_dict()70 entities_positions = prediction_df.set_index("label_position")["position"].to_dict()71 predictions_formatted = {**entities, **entities_probs, **entities_positions}72 73 return predictions_formatted74 else:75 return dict()76 77def clean_prediction(row, feature_name, threshols_dict, clean_functions_dict):78 # Prediction and Probability79 prediction = row[f"pred_{feature_name}"]80 prob = row[f"prob_{feature_name}"]81 82 # Clean and Return Prediction only if the Threshold is lower.83 if prob > threshols_dict[feature_name]:84 clean_function = clean_functions_dict[feature_name]85 prediction_clean = clean_function(prediction)86 return prediction_clean87 else:88 return None89 90surfaces_words_to_omit = ["ha", "hect", "lts", "litros", "mil"]91tower_name_key_words_to_keep = ["torr", "towe"]92 93def has_number(string):94 return bool(re.search(r'\d', string))95 96def contains_multiplication(string):97 # Regular expression pattern to match a multiplication operation98 pattern = r'\b([\d,]+(?:\.\d+)?)\s*(?:\w+\s*)*[xX]\s*([\d,]+(?:\.\d+)?)\s*(?:\w+\s*)*\b'99 100 # Search for the pattern in the string101 match = re.search(pattern, string)102 103 # If a match is found, return True, otherwise False104 if match:105 return True106 else:107 return False108 109def extract_first_number_from_string(text):110 if isinstance(text, str):111 match = re.search(r'\b\d*\.?\d+\b|\d*\.?\d+', text)112 if match:113 start_pos = match.start()114 end_pos = match.end()115 number = int(float(match.group()))116 return number, start_pos, end_pos117 else:118 return None, None, None119 else:120 return None, None, None121 122def get_character(string, index):123 if len(string) > index:124 return string[index]125 else:126 return None127 128def find_valid_comma_separated_number(string):129 # This regular expression matches strings starting with 1 to 3 digits followed by a comma and 3 digits. It ensures no other digits or commas follow or the string ends.130 match = re.match(r'^(\d{1,3},\d{3})(?:[^0-9,]|$)', string)131 if match:132 valid_number = int(match.group(1).replace(",", ""))133 return valid_number134 else:135 return None136 137def extract_surface_from_string(string: str) -> int:138 if isinstance(string, str):139 # 1. Validate if it Contains a Number140 if not(has_number(string)): return None141 142 # 2. Validate if it No Contains Multiplication143 if contains_multiplication(string): return None144 145 # 3. Validate if it No Contains Words to Omit146 if any([word in string.lower() for word in surfaces_words_to_omit]): return None147 148 # 4. Extract First Number149 number, start_pos, end_pos = extract_first_number_from_string(string)150 151 # 5. Extract Valid Comma Separated Number152 if isinstance(number, int):153 if get_character(string, end_pos) == ",": 154 valid_comma_separated_number = find_valid_comma_separated_number(string[start_pos: -1])155 return valid_comma_separated_number156 else:157 return number158 else:159 return None160 else:161 return None162 163def clean_prediction(row, feature_name, threshols_dict, clean_functions_dict):164 # Prediction and Probability165 prediction = row[f"pred_{feature_name}"]166 prob = row[f"prob_{feature_name}"]167 168 # Clean and Return Prediction only if the Threshold is lower.169 if prob > threshols_dict[feature_name]:170 clean_function = clean_functions_dict[feature_name]171 prediction_clean = clean_function(prediction)172 return prediction_clean173 else:174 return None175 176def extract_remodeling_year_from_string(string):177 if isinstance(string, str):178 # 1. Detect 4-digit year179 match = re.search(r'\b\d{4}\b', string)180 if match:181 year_predicted = int(match.group())182 else:183 # 2. Detect quantity of years followed by "year", "years", "anio", "año", or "an"184 match = re.search(r'(\d+) (year|years|anio|año|an|añ)', string.lower(), re.IGNORECASE)185 if match:186 past_years_predicted = int(match.group(1))187 year_predicted = CURRENT_YEAR - past_years_predicted188 else:189 return None190 191 # 3. Detect if it is a valid year192 is_valid_year = (year_predicted <= CURRENT_YEAR) and (YEAR_OF_REMODELING_LIMIT > CURRENT_YEAR - year_predicted)193 return year_predicted if is_valid_year else None194 195 return None196 197def extract_valid_string_left_dotted(string, text, pos):198 if isinstance(string, str):199 # String Position 200 left_pos, rigth_pos = pos201 202 # Verify if the Left Position is not too close to the beginning of the text.203 if left_pos < 5:204 return None205 206 if string[0].isdigit():207 # 1. Take a subtext with 5 more characters to the left of the string.208 sub_text = text[left_pos - 5: rigth_pos]209 210 # 2. If the string has no dots to the left, return the original string.211 if text[left_pos - 1] == ".":212 213 # 3. If the string has a left dot but no preceding digit, return the original string.214 if text[left_pos - 2].isdigit():215 216 # 4. If the string has a left dot, with 3 left digits, and the fourth left value isn't ',', '.', or "''", it returns the new string.217 pattern = r'^(?![\d.,])\D*\d{1,3}\.' + re.escape(string)218 match = re.search(pattern, sub_text)219 if match:220 return match.group(0)221 else:222 return None223 else:224 return string225 else:226 return string227 else:228 return string229 else:230 return None231 232# Cleaning233clean_functions_dict = {234 "SUPERFICIE_TERRAZA": extract_surface_from_string,235 "SUPERFICIE_JARDIN": extract_surface_from_string,236 "SUPERFICIE_TERRENO": extract_surface_from_string,237 "SUPERFICIE_HABITABLE": extract_surface_from_string,238 "SUPERFICIE_BALCON": extract_surface_from_string,239 "AÑO_REMODELACIÓN": extract_remodeling_year_from_string, 240 "NOMBRE_COMPLETO_ARQUITECTO": lambda x: x,241 'NOMBRE_CLUB_GOLF': lambda x: x, 242 'NOMBRE_TORRE': lambda x: x,243 'NOMBRE_CONDOMINIO': lambda x: x,244 'NOMBRE_DESARROLLO': lambda x: x,245}246 247threshols_dict = {248 "SUPERFICIE_TERRAZA": 0.9,249 "SUPERFICIE_JARDIN": 0.9,250 "SUPERFICIE_TERRENO": 0.9,251 "SUPERFICIE_HABITABLE": 0.9,252 "SUPERFICIE_BALCON": 0.9,253 "AÑO_REMODELACIÓN": 0.9,254 "NOMBRE_COMPLETO_ARQUITECTO": 0.9,255 'NOMBRE_CLUB_GOLF': 0.9, 256 'NOMBRE_TORRE': 0.9,257 'NOMBRE_CONDOMINIO': 0.9,258 'NOMBRE_DESARROLLO': 0.9,259}260 261threshols_dict = {262 "SUPERFICIE_BALCON": 0.7697697697697697,263 "SUPERFICIE_TERRAZA": 0.953953953953954,264 "SUPERFICIE_JARDIN": 0.9519519519519519, #idk265 "SUPERFICIE_TERRENO": 0.980980980980981 - 0.05,266 "SUPERFICIE_HABITABLE": 0.978978978978979 - 0.02, #idk if not "SUPERFICIE_HABITABLE": 0.988988988988989,267 "AÑO_REMODELACIÓN": 0.996996996996997 - 0.01,268 "NOMBRE_COMPLETO_ARQUITECTO": 0.8878878878878879,269 "NOMBRE_CLUB_GOLF": 0.8708708708708709, #idk if not "NOMBRE_CLUB_GOLF": 0.9729729729729729,270 "NOMBRE_TORRE": 0.8458458458458459 - 0.04,271 "NOMBRE_CONDOMINIO": 0.965965965965966,272 "NOMBRE_DESARROLLO": 0.9229229229229229273}274 275label_names_dict = {276 'LABEL_0': None, 277 'LABEL_1': 1,278 'LABEL_2': 2, 279 'LABEL_3': 3,280}281BERT_SCORE_LIMIT = 0.980819808198082282 283def extract_max_label_score(probabilities):284 # Find the dictionary with the maximum score285 max_item = max(probabilities, key=lambda x: x['score'])286 # Extract the label and the score287 label = max_item['label']288 score = max_item['score']289 290 return label, score291 292def clean_prediction_bert(label, score):293 if score > BERT_SCORE_LIMIT:294 label_formatted = label_names_dict.get(label, None)295 return label_formatted296 else:297 return None298 299# BERT Inference Config300pipe_config = {301 "batch_size": 8,302 "truncation": True,303 "max_length": 250,304 "add_special_tokens": True,305 "return_all_scores": True,306 "padding": True,307}308 309def generate_answer(text):310 labels = [311 'SUPERFICIE_JARDIN',312 'NOMBRE_CLUB_GOLF',313 'SUPERFICIE_TERRENO',314 'SUPERFICIE_HABITABLE',315 'SUPERFICIE_TERRAZA',316 'NOMBRE_COMPLETO_ARQUITECTO',317 'SUPERFICIE_BALCON',318 'NOMBRE_DESARROLLO',319 'NOMBRE_TORRE',320 'NOMBRE_CONDOMINIO',321 'AÑO_REMODELACIÓN'322 ]323 324 # Clean Text325 text = clean_text(text)326 327 # Inference328 entities = model.predict_entities(text, labels, threshold=0.4)329 330 # Format Prediction Entities331 entities_formatted = format_gliner_predictions(entities)332 333 # Extract valid string left dotted334 feature_surfaces = ['SUPERFICIE_BALCON', 'SUPERFICIE_TERRAZA', 'SUPERFICIE_JARDIN', 'SUPERFICIE_TERRENO', 'SUPERFICIE_HABITABLE']335 for feature_name in feature_surfaces:336 if entities_formatted.get(f"pred_{feature_name}", None) != None:337 entities_formatted[f"pred_{feature_name}"] = extract_valid_string_left_dotted(entities_formatted[f"pred_{feature_name}"], text, entities_formatted[f"pos_{feature_name}"])338 339 # Clean Entities340 entities_names = list({c.replace("pred_", "").replace("prob_", "").replace("pos_", "") for c in list(entities_formatted.keys())})341 entities_cleaned = dict()342 for feature_name in entities_names:343 entity_prediction_cleaned = clean_prediction(entities_formatted, feature_name, threshols_dict, clean_functions_dict)344 if isinstance(entity_prediction_cleaned, str) or isinstance(entity_prediction_cleaned, int):345 entities_cleaned[feature_name] = entity_prediction_cleaned346 347 # BERT Inference348 predictions = pipe([text], **pipe_config)349 350 # Format Prediction351 label, score = extract_max_label_score(predictions[0])352 entities_formatted["NIVELES_CASA"] = label353 entities_formatted["prob_NIVELES_CASA"] = score354 prediction_cleaned = clean_prediction_bert(label, score)355 if isinstance(prediction_cleaned, int):356 entities_cleaned["NIVELES_CASA"] = prediction_cleaned357 358 359 result_json = json.dumps(entities_cleaned, indent = 4, ensure_ascii = False)360 361 return "Clean Result:" + result_json + "\n \n" + "Raw Result:" + json.dumps(entities_formatted, indent = 4, ensure_ascii = False)362 363# Cambiar a entrada de texto364#text_input = gr.inputs.Textbox(lines=15, label="Input Text")365 366iface = gr.Interface(367 fn=generate_answer, 368 inputs="text", 369 outputs="text",370 title="Text Intelligence for Real State",371 description="Input text describing the property."372)373 374iface.launch()375 