beautiful-code/ai_workflows
0
1import pytest2from app.workflows.til.analyse_til import TilCrew # type: ignore3 4 5examples = [6 ("The sun rises in the east.", [7 {"insightful_categorization": 'Low', "factuality_categorization": 'High', "simplicity_categorization": 'High', "grammatical_categorization": 'High'}]),8 ("* Quantization is the process of reducing the size of LLM models by reducing the underlying weights.\n"9 "* In quantization the weights are reduced by scaling up the datatypes from a datatype that takes smaller space to a data type that takes a larger space, this is also known as downcasting for example downcasting from int8 to float32.\n"10 "* Advantages: takes lesser space and increases compute speed.\n"11 "* Disadvantages: Answers are less precise because of the loss of precision in the LLM model weights.\n", [12 {"insightful_categorization": 'Meidum', "factuality_categorization": 'High',13 "simplicity_categorization": 'High', "grammatical_categorization": 'High'},14 {"insightful_categorization": 'High', "factuality_categorization": 'Low',15 "simplicity_categorization": 'High', "grammatical_categorization": 'High'},16 {"insightful_categorization": 'High', "factuality_categorization": 'High',17 "simplicity_categorization": 'High', "grammatical_categorization": 'High'},18 {"insightful_categorization": 'High', "factuality_categorization": 'High',19 "simplicity_categorization": 'High', "grammatical_categorization": 'High'},20 ]),21]22 23 24@pytest.mark.parametrize("input_text, expected_categorizations", examples)25def test_llm_evaluation(input_text, expected_categorizations):26 til_crew = TilCrew()27 til_crew.content = input_text28 til_crew._gather_feedback()29 response = til_crew.feedback_results30 31 for idx, feedback in enumerate(response):32 assert feedback["insightful_categorization"] == pytest.approx(33 expected_categorizations[idx]["insightful_categorization"], abs=2.0)34 assert feedback["factuality_categorization"] == pytest.approx(35 expected_categorizations[idx]["factuality_categorization"], abs=2.0)36 assert feedback["simplicity_categorization"] == pytest.approx(37 expected_categorizations[idx]["simplicity_categorization"], abs=2.0)38 assert feedback["grammatical_categorization"] == pytest.approx(39 expected_categorizations[idx]["grammatical_categorization"], abs=2.0)40 