Team Ai
Apppublic

beautiful-code/ai_workflows

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes
til_test.py40 linesDownload Raw Back to tests
1import pytest2from app.workflows.til.analyse_til import TilCrew  # type: ignore3 4 5examples = [6    ("The sun rises in the east.", [7     {"insightful_categorization": 'Low', "factuality_categorization": 'High', "simplicity_categorization": 'High', "grammatical_categorization": 'High'}]),8    ("* Quantization is the process of reducing the size of LLM models by reducing the underlying weights.\n"9     "* In quantization the weights are reduced by scaling up the datatypes from a datatype that takes smaller space to a data type that takes a larger space, this is also known as downcasting for example downcasting from int8 to float32.\n"10     "* Advantages: takes lesser space and increases compute speed.\n"11     "* Disadvantages: Answers are less precise because of the loss of precision in the LLM model weights.\n", [12        {"insightful_categorization": 'Meidum', "factuality_categorization": 'High',13            "simplicity_categorization": 'High', "grammatical_categorization": 'High'},14        {"insightful_categorization": 'High', "factuality_categorization": 'Low',15            "simplicity_categorization": 'High', "grammatical_categorization": 'High'},16        {"insightful_categorization": 'High', "factuality_categorization": 'High',17            "simplicity_categorization": 'High', "grammatical_categorization": 'High'},18        {"insightful_categorization": 'High', "factuality_categorization": 'High',19            "simplicity_categorization": 'High', "grammatical_categorization": 'High'},20    ]),21]22 23 24@pytest.mark.parametrize("input_text, expected_categorizations", examples)25def test_llm_evaluation(input_text, expected_categorizations):26    til_crew = TilCrew()27    til_crew.content = input_text28    til_crew._gather_feedback()29    response = til_crew.feedback_results30 31    for idx, feedback in enumerate(response):32        assert feedback["insightful_categorization"] == pytest.approx(33            expected_categorizations[idx]["insightful_categorization"], abs=2.0)34        assert feedback["factuality_categorization"] == pytest.approx(35            expected_categorizations[idx]["factuality_categorization"], abs=2.0)36        assert feedback["simplicity_categorization"] == pytest.approx(37            expected_categorizations[idx]["simplicity_categorization"], abs=2.0)38        assert feedback["grammatical_categorization"] == pytest.approx(39            expected_categorizations[idx]["grammatical_categorization"], abs=2.0)40