Team Ai
Apppublic

TangibleAI/mathtext-fastapi

sourceHugging Faceagpl-3.0updated 3y agoView on Hugging Face
1likes
1import re2 3from collections.abc import Mapping4from logging import getLogger5import datetime as dt6from dateutil.parser import isoparse7 8from fuzzywuzzy import fuzz9from fuzzywuzzy import process10from mathtext_fastapi.intent_classification import predict_message_intent11from mathtext_fastapi.logging import prepare_message_data_for_logging12from mathtext.sentiment import sentiment13from mathtext.text2int import text2int, TOKENS2INT_ERROR_INT14 15log = getLogger(__name__)16 17PAYLOAD_VALUE_TYPES = {18    'author_id': str,19    'author_type': str,20    'contact_uuid': str,21    'message_body': str,22    'message_direction': str,23    'message_id': str,24    'message_inserted_at': str,25    'message_updated_at': str,26    }27 28 29def build_nlu_response_object(nlu_type, data, confidence):30    """ Turns nlu results into an object to send back to Turn.io31    Inputs32    - nlu_type: str - the type of nlu run (integer or sentiment-analysis)33    - data: str/int - the student message34    - confidence: - the nlu confidence score (sentiment) or '' (integer)35 36    >>> build_nlu_response_object('integer', 8, 0)37    {'type': 'integer', 'data': 8, 'confidence': 0}38 39    >>> build_nlu_response_object('sentiment', 'POSITIVE', 0.99)40    {'type': 'sentiment', 'data': 'POSITIVE', 'confidence': 0.99}41    """42    return {43        'type': nlu_type,44        'data': data,45        'confidence': confidence46        }47 48 49# def test_for_float_or_int(message_data, message_text):50#     nlu_response = {}51#     if type(message_text) == int or type(message_text) == float:52#         nlu_response = build_nlu_response_object('integer', message_text, '')53#         prepare_message_data_for_logging(message_data, nlu_response)54#     return nlu_response55 56 57def test_for_number_sequence(message_text_arr, message_data, message_text):58    """ Determines if the student's message is a sequence of numbers59 60    >>> test_for_number_sequence(['1','2','3'], {"author_id": "57787919091", "author_type": "OWNER", "contact_uuid": "df78gsdf78df", "message_body": "I am tired", "message_direction": "inbound", "message_id": "dfgha789789ag9ga", "message_inserted_at": "2023-01-10T02:37:28.487319Z", "message_updated_at": "2023-01-10T02:37:28.487319Z"}, '1, 2, 3')61    {'type': 'integer', 'data': '1,2,3', 'confidence': 0}62 63    >>> test_for_number_sequence(['a','b','c'], {"author_id": "57787919091", "author_type": "OWNER", "contact_uuid": "df78gsdf78df", "message_body": "I am tired", "message_direction": "inbound", "message_id": "dfgha789789ag9ga", "message_inserted_at": "2023-01-10T02:37:28.487319Z", "message_updated_at": "2023-01-10T02:37:28.487319Z"}, 'a, b, c')64    {}65    """66    nlu_response = {}67    if all(ele.isdigit() for ele in message_text_arr):68        nlu_response = build_nlu_response_object(69            'integer',70            ','.join(message_text_arr),71            072        )73        prepare_message_data_for_logging(message_data, nlu_response)74    return nlu_response75 76 77def run_text2int_on_each_list_item(message_text_arr):78    """ Attempts to convert each list item to an integer79 80    Input81    - message_text_arr: list - a set of text extracted from the student message82 83    Output84    - student_response_arr: list - a set of integers (32202 for error code)85 86    >>> run_text2int_on_each_list_item(['1','2','3'])87    [1, 2, 3]88    """89    student_response_arr = []90    for student_response in message_text_arr:91        int_api_resp = text2int(student_response.lower())92        student_response_arr.append(int_api_resp)93    return student_response_arr94 95 96def run_sentiment_analysis(message_text):97    """ Evaluates the sentiment of a student message98 99    >>> run_sentiment_analysis("I am tired")100    [{'label': 'NEGATIVE', 'score': 0.9997807145118713}]101 102    >>> run_sentiment_analysis("I am full of joy")103    [{'label': 'POSITIVE', 'score': 0.999882698059082}]104    """105    # TODO: Add intent labelling here106    # TODO: Add logic to determine whether intent labeling or sentiment analysis is more appropriate (probably default to intent labeling)107    return sentiment(message_text)108 109 110def run_intent_classification(message_text):111    """ Process a student's message using basic fuzzy text comparison112 113    >>> run_intent_classification("exit")114    {'type': 'intent', 'data': 'exit', 'confidence': 1.0}115    >>> run_intent_classification("exi")     116    {'type': 'intent', 'data': 'exit', 'confidence': 0.86}117    >>> run_intent_classification("eas")118    {'type': 'intent', 'data': '', 'confidence': 0}119    >>> run_intent_classification("hard")120    {'type': 'intent', 'data': '', 'confidence': 0}121    >>> run_intent_classification("hardier") 122    {'type': 'intent', 'data': 'harder', 'confidence': 0.92}123    """124    label = ''125    ratio = 0126    nlu_response = {'type': 'intent', 'data': label, 'confidence': ratio}127    keywords = [128        'easier',129        'exit',130        'harder',131        'hint',132        'next',133        'stop',134        'tired',135        'tomorrow',136        'finished',137        'help',138        'easier',139        'easy',140        'support',141        'skip',142        'menu'143    ]144 145    try:146        tokens = re.findall(r"[-a-zA-Z'_]+", message_text.lower())147    except AttributeError:148        tokens = ''149 150    for keyword in keywords:151        try:152            tok, score = process.extractOne(keyword, tokens, scorer=fuzz.ratio)153        except:154            score = 0155 156        if score > 80:157            nlu_response['data'] = keyword158            nlu_response['confidence'] = score159    160    return nlu_response161 162 163def payload_is_valid(payload_object):164    """165    >>> payload_is_valid({'author_id': '+5555555', 'author_type': 'OWNER', 'contact_uuid': '3246-43ad-faf7qw-zsdhg-dgGdg', 'message_body': 'thirty one', 'message_direction': 'inbound', 'message_id': 'SDFGGwafada-DFASHA4aDGA', 'message_inserted_at': '2022-07-05T04:00:34.03352Z', 'message_updated_at': '2023-04-06T10:08:23.745072Z'})166    True167 168    >>> payload_is_valid({"author_id": "@event.message._vnd.v1.chat.owner", "author_type": "@event.message._vnd.v1.author.type", "contact_uuid": "@event.message._vnd.v1.chat.contact_uuid", "message_body": "@event.message.text.body", "message_direction": "@event.message._vnd.v1.direction", "message_id": "@event.message.id", "message_inserted_at": "@event.message._vnd.v1.chat.inserted_at", "message_updated_at": "@event.message._vnd.v1.chat.updated_at"})169    False170    """171    try:172        isinstance(173            isoparse(payload_object.get('message_inserted_at','')),174            dt.datetime175        )176        isinstance(177            isoparse(payload_object.get('message_updated_at','')),178            dt.datetime179        )180    except ValueError:181        return False182    return (183        isinstance(payload_object, Mapping) and184        isinstance(payload_object.get('author_id'), str) and185        isinstance(payload_object.get('author_type'), str) and186        isinstance(payload_object.get('contact_uuid'), str) and187        isinstance(payload_object.get('message_body'), str) and188        isinstance(payload_object.get('message_direction'), str) and189        isinstance(payload_object.get('message_id'), str) and190        isinstance(payload_object.get('message_inserted_at'), str) and191        isinstance(payload_object.get('message_updated_at'), str)    192    )193 194 195def log_payload_errors(payload_object):196    errors = []197    try:198        assert isinstance(payload_object, Mapping)199    except Exception as e:200        log.error(f'Invalid HTTP request payload object: {e}')201        errors.append(e)202    for k, typ in PAYLOAD_VALUE_TYPES.items():203        try:204            assert isinstance(payload_object.get(k), typ)205        except Exception as e:206            log.error(f'Invalid HTTP request payload object: {e}')207            errors.append(e)208    try:209        assert isinstance(210            dt.datetime.fromisoformat(payload_object.get('message_inserted_at')),211            dt.datetime212        )213    except Exception as e:214        log.error(f'Invalid HTTP request payload object: {e}')215        errors.append(e)216    try: 217        isinstance(218            dt.datetime.fromisoformat(payload_object.get('message_updated_at')),219            dt.datetime220        )221    except Exception as e:222        log.error(f'Invalid HTTP request payload object: {e}')223        errors.append(e)224    return errors225 226 227def evaluate_message_with_nlu(message_data):228    """ Process a student's message using NLU functions and send the result229    230    >>> evaluate_message_with_nlu({"author_id": "57787919091", "author_type": "OWNER", "contact_uuid": "df78gsdf78df", "message_body": "8", "message_direction": "inbound", "message_id": "dfgha789789ag9ga", "message_inserted_at": "2023-01-10T02:37:28.487319Z", "message_updated_at": "2023-01-10T02:37:28.487319Z"})231    {'type': 'integer', 'data': 8, 'confidence': 0}232 233    >>> evaluate_message_with_nlu({"author_id": "57787919091", "author_type": "OWNER", "contact_uuid": "df78gsdf78df", "message_body": "I am tired", "message_direction": "inbound", "message_id": "dfgha789789ag9ga", "message_inserted_at": "2023-01-10T02:37:28.487319Z", "message_updated_at": "2023-01-10T02:37:28.487319Z"})234    {'type': 'sentiment', 'data': 'NEGATIVE', 'confidence': 0.9997807145118713}235    """236    # Keeps system working with two different inputs - full and filtered @event object237    # Call validate payload238    log.info(f'Starting evaluate message: {message_data}')239 240    if not payload_is_valid(message_data):241        log_payload_errors(message_data)242        return {'type': 'error', 'data': TOKENS2INT_ERROR_INT, 'confidence': 0}243 244    try:245        message_text = str(message_data.get('message_body', ''))246    except:247        log.error(f'Invalid request payload: {message_data}')248        # use python logging system to do this//249        return {'type': 'error', 'data': TOKENS2INT_ERROR_INT, 'confidence': 0}250 251    # Run intent classification only for keywords252    intent_api_response = run_intent_classification(message_text)253    if intent_api_response['data']:254        prepare_message_data_for_logging(message_data, intent_api_response)255        return intent_api_response256 257    number_api_resp = text2int(message_text.lower())258 259    if number_api_resp == TOKENS2INT_ERROR_INT:260        # Run intent classification with logistic regression model261        predicted_label = predict_message_intent(message_text)262        if predicted_label['confidence'] > 0.01:263            nlu_response = predicted_label264        else:265            # Run sentiment analysis266            sentiment_api_resp = sentiment(message_text)267            nlu_response = build_nlu_response_object(268                'sentiment',269                sentiment_api_resp[0]['label'],270                sentiment_api_resp[0]['score']271            )272    else:273        nlu_response = build_nlu_response_object(274            'integer',275            number_api_resp,276            0277        )278 279    prepare_message_data_for_logging(message_data, nlu_response)280    return nlu_response281