Team Ai
Apppublic

Agents-MCP-Hackathon/Python-Code-to-Diagram-Generator-MCP

sourceHugging Facemitupdated 1y agoView on Hugging Face
6likes
string_processing.py164 linesDownload Raw Back to examples
1"""2String processing pipeline functions for testing function analysis.3"""4 5import re6from typing import List7 8 9def normalize_whitespace(text):10    """Normalize whitespace by removing extra spaces and newlines."""11    # Replace multiple whitespace with single space12    text = re.sub(r'\s+', ' ', text)13    # Strip leading and trailing whitespace14    return text.strip()15 16 17def remove_special_characters(text, keep_chars=""):18    """Remove special characters, optionally keeping specified characters."""19    # Keep alphanumeric, spaces, and specified characters20    pattern = fr"[^a-zA-Z0-9\s{re.escape(keep_chars)}]"21    return re.sub(pattern, '', text)22 23 24def convert_to_lowercase(text):25    """Convert text to lowercase."""26    return text.lower()27 28 29def remove_stopwords(text, stopwords=None):30    """Remove common stopwords from text."""31    if stopwords is None:32        stopwords = {33            'the', 'a', 'an', 'and', 'or', 'but', 'in', 'on', 'at', 'to', 34            'for', 'of', 'with', 'by', 'is', 'are', 'was', 'were', 'be',35            'been', 'being', 'have', 'has', 'had', 'do', 'does', 'did',36            'will', 'would', 'could', 'should', 'may', 'might', 'must'37        }38    39    words = text.split()40    filtered_words = [word for word in words if word.lower() not in stopwords]41    return ' '.join(filtered_words)42 43 44def extract_keywords(text, min_length=3):45    """Extract keywords (words longer than min_length)."""46    words = text.split()47    keywords = [word for word in words if len(word) >= min_length]48    return keywords49 50 51def count_word_frequency(text):52    """Count frequency of each word in text."""53    words = text.split()54    frequency = {}55    for word in words:56        frequency[word] = frequency.get(word, 0) + 157    return frequency58 59 60def capitalize_words(text, exceptions=None):61    """Capitalize first letter of each word, with exceptions."""62    if exceptions is None:63        exceptions = {'a', 'an', 'the', 'and', 'or', 'but', 'in', 'on', 'at', 'to', 'for', 'of', 'with', 'by'}64    65    words = text.split()66    capitalized = []67    68    for i, word in enumerate(words):69        if i == 0 or word.lower() not in exceptions:70            capitalized.append(word.capitalize())71        else:72            capitalized.append(word.lower())73    74    return ' '.join(capitalized)75 76 77def truncate_text(text, max_length=100, suffix="..."):78    """Truncate text to specified length with suffix."""79    if len(text) <= max_length:80        return text81    82    truncated = text[:max_length - len(suffix)]83    # Try to break at last complete word84    last_space = truncated.rfind(' ')85    if last_space > max_length * 0.8:  # If we can break at a word boundary86        truncated = truncated[:last_space]87    88    return truncated + suffix89 90 91def text_processing_pipeline(text, operations=None):92    """Process text through a pipeline of operations."""93    if operations is None:94        operations = [95            'normalize_whitespace',96            'remove_special_characters', 97            'convert_to_lowercase',98            'remove_stopwords'99        ]100    101    # Map operation names to functions102    operation_map = {103        'normalize_whitespace': normalize_whitespace,104        'remove_special_characters': remove_special_characters,105        'convert_to_lowercase': convert_to_lowercase,106        'remove_stopwords': remove_stopwords,107        'capitalize_words': capitalize_words,108        'truncate_text': truncate_text109    }110    111    result = text112    processing_steps = []113    114    for operation in operations:115        if operation in operation_map:116            before = result117            result = operation_map[operation](result)118            processing_steps.append({119                'operation': operation,120                'before': before[:50] + "..." if len(before) > 50 else before,121                'after': result[:50] + "..." if len(result) > 50 else result122            })123    124    return result, processing_steps125 126 127def analyze_text_statistics(text):128    """Analyze various statistics about the text."""129    words = text.split()130    131    stats = {132        'character_count': len(text),133        'word_count': len(words),134        'sentence_count': len(re.findall(r'[.!?]+', text)),135        'average_word_length': sum(len(word) for word in words) / len(words) if words else 0,136        'longest_word': max(words, key=len) if words else "",137        'shortest_word': min(words, key=len) if words else ""138    }139    140    return stats141 142 143if __name__ == "__main__":144    sample_text = """145    This is a SAMPLE text with various   formatting issues!!! 146    It has multiple    spaces, special @#$% characters, and 147    needs some serious cleaning & processing...148    """149    150    print("Original text:")151    print(repr(sample_text))152    153    processed_text, steps = text_processing_pipeline(sample_text)154    155    print("\nProcessing steps:")156    for step in steps:157        print(f"After {step['operation']}:")158        print(f"  {step['after']}")159    160    print(f"\nFinal result: {processed_text}")161    162    stats = analyze_text_statistics(processed_text)163    print(f"\nText statistics: {stats}")164