Team Ai
Apppublic

chris32/Text-Intelligence-Real-State

sourceHugging Faceapache-2.0updated 2y agoView on Hugging Face
1likes
app.py375 linesDownload Raw Back to root
1# Datetime2import datetime3# Manipulate4import re5import json6import pandas as pd7# App8import gradio as gr9# GLiNER Model10from gliner import GLiNER11# Transformers12from transformers import pipeline13 14# Load GLiNER Model15model = GLiNER.from_pretrained("chris32/gliner_multi_pii_real_state-v2")16model.eval()17 18# BERT Model19model_name = "chris32/distilbert-base-spanish-uncased-finetuned-text-intelligence"20pipe = pipeline(model = model_name, device = "cpu")21 22# Global Variables: For Post Cleaning Inferences23YEAR_OF_REMODELING_LIMIT = 10024CURRENT_YEAR = int(datetime.date.today().year)25SCORE_LIMIT_SIMILARITY_NAMES = 7026 27def clean_text(text):28    # Replace HTML line breaks with the specified character29    replacement_char = " # "30    text = re.sub(r'<br\s*\/?>', replacement_char, text)31    32    # Remove HTML tags and special characters33    cleaned_text = re.sub(r'<[^>]*>', '', text)34    cleaned_text = re.sub(r'&nbsp;', ' ', cleaned_text)35    cleaned_text = re.sub(r'&amp;', '&', cleaned_text)36    37    # Drop punctuation marks38    #regex = '[\\!\\"\\#\\$\\%\\&\\\'\\(\\)\\*\\+\\,\\-\\.\\/\\:\\;\\<\\=\\>\\?\\@\\[\\\\\\]\\^_\\`\\{\\|\\}\\~]'39    #cleaned_text = re.sub(regex , ' ', cleaned_text)40    41    # Replace multiple spaces with a single one42    cleaned_text = re.sub(r'\s+', ' ', cleaned_text)43 44    # Remove leading and trailing spaces45    cleaned_text = cleaned_text.strip()46    47    # Replace Duplicated "." and ","48    cleaned_text = cleaned_text.replace("..", ".").replace(",,", ",")49    50    return cleaned_text51 52def format_gliner_predictions(prediction):53    if len(prediction) > 0:    54        # Select the Entity value with the Greater Score for each Entity Name55        prediction_df = pd.DataFrame(prediction)\56                          .sort_values("score", ascending = False)\57                          .drop_duplicates(subset = "label", keep = "first")58 59        # Add Position Column60        prediction_df["position"] = prediction_df.apply(lambda x: (x["start"], x["end"]) ,axis = 1)61 62        # Add Columns Label for Text and Probability63        prediction_df["label_text"] = prediction_df["label"].apply(lambda x: f"pred_{x}")64        prediction_df["label_prob"] = prediction_df["label"].apply(lambda x: f"prob_{x}")65        prediction_df["label_position"] = prediction_df["label"].apply(lambda x: f"pos_{x}")66 67        # Format Predictions68        entities = prediction_df.set_index("label_text")["text"].to_dict()69        entities_probs = prediction_df.set_index("label_prob")["score"].to_dict()70        entities_positions = prediction_df.set_index("label_position")["position"].to_dict()71        predictions_formatted = {**entities, **entities_probs, **entities_positions}72 73        return predictions_formatted74    else:75        return dict()76    77def clean_prediction(row, feature_name, threshols_dict, clean_functions_dict):78    # Prediction and Probability79    prediction = row[f"pred_{feature_name}"]80    prob = row[f"prob_{feature_name}"]81    82    # Clean and Return Prediction only if the Threshold is lower.83    if prob > threshols_dict[feature_name]:84        clean_function = clean_functions_dict[feature_name]85        prediction_clean = clean_function(prediction)86        return prediction_clean87    else:88        return None89    90surfaces_words_to_omit = ["ha", "hect", "lts", "litros", "mil"]91tower_name_key_words_to_keep = ["torr", "towe"]92 93def has_number(string):94    return bool(re.search(r'\d', string))95 96def contains_multiplication(string):97    # Regular expression pattern to match a multiplication operation98    pattern = r'\b([\d,]+(?:\.\d+)?)\s*(?:\w+\s*)*[xX]\s*([\d,]+(?:\.\d+)?)\s*(?:\w+\s*)*\b'99    100    # Search for the pattern in the string101    match = re.search(pattern, string)102    103    # If a match is found, return True, otherwise False104    if match:105        return True106    else:107        return False108 109def extract_first_number_from_string(text):110    if isinstance(text, str):111        match = re.search(r'\b\d*\.?\d+\b|\d*\.?\d+', text)112        if match:113            start_pos = match.start()114            end_pos = match.end()115            number = int(float(match.group()))116            return number, start_pos, end_pos117        else:118            return None, None, None119    else:120        return None, None, None121    122def get_character(string, index):123    if len(string) > index:124        return string[index]125    else:126        return None127    128def find_valid_comma_separated_number(string):129    # This regular expression matches strings starting with 1 to 3 digits followed by a comma and 3 digits. It ensures no other digits or commas follow or the string ends.130    match = re.match(r'^(\d{1,3},\d{3})(?:[^0-9,]|$)', string)131    if match:132        valid_number = int(match.group(1).replace(",", ""))133        return valid_number134    else:135        return None136 137def extract_surface_from_string(string: str) -> int:138    if isinstance(string, str):139        # 1. Validate if it Contains a Number140        if not(has_number(string)): return None141 142        # 2. Validate if it No Contains Multiplication143        if contains_multiplication(string): return None144 145        # 3. Validate if it No Contains Words to Omit146        if any([word in string.lower() for word in surfaces_words_to_omit]): return None147 148        # 4. Extract First Number149        number, start_pos, end_pos = extract_first_number_from_string(string)150 151        # 5. Extract Valid Comma Separated Number152        if isinstance(number, int):153            if get_character(string, end_pos) == ",": 154                valid_comma_separated_number = find_valid_comma_separated_number(string[start_pos: -1])155                return valid_comma_separated_number156            else:157                return number158        else:159            return None160    else:161        return None162    163def clean_prediction(row, feature_name, threshols_dict, clean_functions_dict):164    # Prediction and Probability165    prediction = row[f"pred_{feature_name}"]166    prob = row[f"prob_{feature_name}"]167    168    # Clean and Return Prediction only if the Threshold is lower.169    if prob > threshols_dict[feature_name]:170        clean_function = clean_functions_dict[feature_name]171        prediction_clean = clean_function(prediction)172        return prediction_clean173    else:174        return None175 176def extract_remodeling_year_from_string(string):177    if isinstance(string, str):178        # 1. Detect 4-digit year179        match = re.search(r'\b\d{4}\b', string)180        if match:181            year_predicted = int(match.group())182        else:183            # 2. Detect quantity of years followed by "year", "years", "anio", "año", or "an"184            match = re.search(r'(\d+) (year|years|anio|año|an|añ)', string.lower(), re.IGNORECASE)185            if match:186                past_years_predicted = int(match.group(1))187                year_predicted = CURRENT_YEAR - past_years_predicted188            else:189                return None190        191        # 3. Detect if it is a valid year192        is_valid_year = (year_predicted <= CURRENT_YEAR) and (YEAR_OF_REMODELING_LIMIT > CURRENT_YEAR - year_predicted)193        return year_predicted if is_valid_year else None194        195    return None196 197def extract_valid_string_left_dotted(string, text, pos):198    if isinstance(string, str):199        # String Position 200        left_pos, rigth_pos = pos201 202        # Verify if the Left Position is not too close to the beginning of the text.203        if left_pos < 5:204            return None205 206        if string[0].isdigit():207            # 1. Take a subtext with 5 more characters to the left of the string.208            sub_text = text[left_pos - 5: rigth_pos]209 210            # 2. If the string has no dots to the left, return the original string.211            if text[left_pos - 1] == ".":212 213                # 3. If the string has a left dot but no preceding digit, return the original string.214                if text[left_pos - 2].isdigit():215 216                    # 4. If the string has a left dot, with 3 left digits, and the fourth left value isn't ',', '.', or "''", it returns the new string.217                    pattern = r'^(?![\d.,])\D*\d{1,3}\.' + re.escape(string)218                    match = re.search(pattern, sub_text)219                    if match:220                        return match.group(0)221                    else:222                        return None223                else:224                    return string225            else:226                return string227        else:228            return string229    else:230        return None231    232# Cleaning233clean_functions_dict = {234    "SUPERFICIE_TERRAZA": extract_surface_from_string,235    "SUPERFICIE_JARDIN": extract_surface_from_string,236    "SUPERFICIE_TERRENO": extract_surface_from_string,237    "SUPERFICIE_HABITABLE": extract_surface_from_string,238    "SUPERFICIE_BALCON": extract_surface_from_string,239    "AÑO_REMODELACIÓN": extract_remodeling_year_from_string, 240    "NOMBRE_COMPLETO_ARQUITECTO": lambda x: x,241    'NOMBRE_CLUB_GOLF': lambda x: x, 242    'NOMBRE_TORRE': lambda x: x,243    'NOMBRE_CONDOMINIO': lambda x: x,244    'NOMBRE_DESARROLLO': lambda x: x,245}246 247threshols_dict = {248    "SUPERFICIE_TERRAZA": 0.9,249    "SUPERFICIE_JARDIN": 0.9,250    "SUPERFICIE_TERRENO": 0.9,251    "SUPERFICIE_HABITABLE": 0.9,252    "SUPERFICIE_BALCON": 0.9,253    "AÑO_REMODELACIÓN": 0.9,254    "NOMBRE_COMPLETO_ARQUITECTO": 0.9,255    'NOMBRE_CLUB_GOLF': 0.9, 256    'NOMBRE_TORRE': 0.9,257    'NOMBRE_CONDOMINIO': 0.9,258    'NOMBRE_DESARROLLO': 0.9,259}260 261threshols_dict = {262    "SUPERFICIE_BALCON": 0.7697697697697697,263    "SUPERFICIE_TERRAZA": 0.953953953953954,264    "SUPERFICIE_JARDIN": 0.9519519519519519, #idk265    "SUPERFICIE_TERRENO": 0.980980980980981 - 0.05,266    "SUPERFICIE_HABITABLE": 0.978978978978979 - 0.02, #idk if not "SUPERFICIE_HABITABLE": 0.988988988988989,267    "AÑO_REMODELACIÓN": 0.996996996996997 - 0.01,268    "NOMBRE_COMPLETO_ARQUITECTO": 0.8878878878878879,269    "NOMBRE_CLUB_GOLF": 0.8708708708708709, #idk if not "NOMBRE_CLUB_GOLF": 0.9729729729729729,270    "NOMBRE_TORRE": 0.8458458458458459 - 0.04,271    "NOMBRE_CONDOMINIO": 0.965965965965966,272    "NOMBRE_DESARROLLO": 0.9229229229229229273}274 275label_names_dict = {276    'LABEL_0': None, 277    'LABEL_1': 1,278    'LABEL_2': 2, 279    'LABEL_3': 3,280}281BERT_SCORE_LIMIT = 0.980819808198082282 283def extract_max_label_score(probabilities):284    # Find the dictionary with the maximum score285    max_item = max(probabilities, key=lambda x: x['score'])286    # Extract the label and the score287    label = max_item['label']288    score = max_item['score']289 290    return label, score291 292def clean_prediction_bert(label, score):293    if score > BERT_SCORE_LIMIT:294        label_formatted = label_names_dict.get(label, None)295        return  label_formatted296    else:297        return None298    299# BERT Inference Config300pipe_config = {301    "batch_size": 8,302    "truncation": True,303    "max_length": 250,304    "add_special_tokens": True,305    "return_all_scores": True,306    "padding": True,307}308 309def generate_answer(text):310    labels = [311    'SUPERFICIE_JARDIN',312    'NOMBRE_CLUB_GOLF',313    'SUPERFICIE_TERRENO',314    'SUPERFICIE_HABITABLE',315    'SUPERFICIE_TERRAZA',316    'NOMBRE_COMPLETO_ARQUITECTO',317    'SUPERFICIE_BALCON',318    'NOMBRE_DESARROLLO',319    'NOMBRE_TORRE',320    'NOMBRE_CONDOMINIO',321    'AÑO_REMODELACIÓN'322    ]323 324    # Clean Text325    text = clean_text(text)326    327    # Inference328    entities = model.predict_entities(text, labels, threshold=0.4)329 330    # Format Prediction Entities331    entities_formatted = format_gliner_predictions(entities)332 333    # Extract valid string left dotted334    feature_surfaces = ['SUPERFICIE_BALCON', 'SUPERFICIE_TERRAZA', 'SUPERFICIE_JARDIN', 'SUPERFICIE_TERRENO', 'SUPERFICIE_HABITABLE']335    for feature_name in feature_surfaces:336        if entities_formatted.get(f"pred_{feature_name}", None) != None:337           entities_formatted[f"pred_{feature_name}"] = extract_valid_string_left_dotted(entities_formatted[f"pred_{feature_name}"], text, entities_formatted[f"pos_{feature_name}"])338 339    # Clean Entities340    entities_names = list({c.replace("pred_", "").replace("prob_", "").replace("pos_", "") for c in list(entities_formatted.keys())})341    entities_cleaned = dict()342    for feature_name in entities_names:343        entity_prediction_cleaned = clean_prediction(entities_formatted, feature_name, threshols_dict, clean_functions_dict)344        if isinstance(entity_prediction_cleaned, str) or isinstance(entity_prediction_cleaned, int):345            entities_cleaned[feature_name] = entity_prediction_cleaned346    347    # BERT Inference348    predictions = pipe([text], **pipe_config)349 350    # Format Prediction351    label, score = extract_max_label_score(predictions[0])352    entities_formatted["NIVELES_CASA"] = label353    entities_formatted["prob_NIVELES_CASA"] = score354    prediction_cleaned = clean_prediction_bert(label, score)355    if isinstance(prediction_cleaned, int):356        entities_cleaned["NIVELES_CASA"] = prediction_cleaned357    358 359    result_json = json.dumps(entities_cleaned, indent = 4, ensure_ascii = False)360 361    return "Clean Result:" + result_json + "\n \n" + "Raw Result:" + json.dumps(entities_formatted, indent = 4, ensure_ascii = False)362 363# Cambiar a entrada de texto364#text_input = gr.inputs.Textbox(lines=15, label="Input Text")365 366iface = gr.Interface(367    fn=generate_answer, 368    inputs="text", 369    outputs="text",370    title="Text Intelligence for Real State",371    description="Input text describing the property."372)373 374iface.launch()375