Team Ai
Apppublic

abdulshakur/YT-TranscriptSegmenter

sourceHugging Facemitupdated 2y agoView on Hugging Face
0likes
app.py195 linesDownload Raw Back to root
1import gradio as gr2import nltk3import re4from nltk.tokenize import sent_tokenize5 6# Make sure NLTK data is downloaded correctly7import os8os.environ['NLTK_DATA'] = '/home/user/nltk_data'9nltk.download('punkt', quiet=True, download_dir='/home/user/nltk_data')10 11# Make sure the data is available12try:13    nltk.data.find('tokenizers/punkt')14    print("NLTK punkt tokenizer found successfully!")15except LookupError as e:16    print(f"Error finding punkt tokenizer: {e}")17    # Try a more explicit download18    nltk.download('punkt', download_dir='/home/user/nltk_data')19    print("Attempted explicit download of punkt")20 21def count_tokens(text):22    """23    Estimate the number of tokens in a text.24    This is a rough approximation based on counting words and punctuation.25    """26    # Split on whitespace and keep punctuation as tokens27    try:28        words = re.findall(r'\b\w+\b|[.,!?;:]', text)29        return len(words)30    except Exception as e:31        print(f"Error counting tokens: {e}")32        # Fallback to a simpler method33        return len(text.split())34 35def segment_transcript(transcript, max_segment_length=1500, smart_boundaries=True):36    """37    Segments a transcript into smaller chunks for processing.38    39    Args:40        transcript: The full transcript text41        max_segment_length: Maximum length of each segment in characters42        smart_boundaries: Whether to use sentence boundaries for smarter segmentation43    44    Returns:45        A list of segments46    """47    if not transcript or transcript.strip() == "":48        return []49    50    # Clean up the transcript by normalizing whitespace51    transcript = re.sub(r'\s+', ' ', transcript).strip()52    53    if smart_boundaries:54        try:55            # Use sentence tokenization for smarter segmentation56            sentences = sent_tokenize(transcript)57            print(f"Successfully tokenized transcript into {len(sentences)} sentences")58        except Exception as e:59            print(f"Error during sentence tokenization: {e}")60            print("Falling back to simple segmentation")61            # Fall back to simple segmentation62            smart_boundaries = False63    64    if smart_boundaries:65        segments = []66        current_segment = ""67        current_token_count = 068        estimated_token_limit = max_segment_length  # Characters as rough approximation69        70        for sentence in sentences:71            sentence_token_count = count_tokens(sentence)72            73            if current_token_count + sentence_token_count <= estimated_token_limit:74                current_segment += sentence + " "75                current_token_count += sentence_token_count76            else:77                if current_segment:78                    segments.append(current_segment.strip())79                current_segment = sentence + " "80                current_token_count = sentence_token_count81        82        if current_segment:  # Add the last segment if it exists83            segments.append(current_segment.strip())84    else:85        # Simple character-based chunking without respecting sentence boundaries86        segments = []87        for i in range(0, len(transcript), max_segment_length):88            segments.append(transcript[i:i + max_segment_length])89    90    # Add segment numbers for easy reference91    numbered_segments = [f"Segment {i+1}/{len(segments)}:\n{segment}" 92                         for i, segment in enumerate(segments)]93    94    print(f"Created {len(numbered_segments)} segments")95    return numbered_segments96 97def process_transcript(transcript, max_length, use_smart_boundaries):98    """Main function that processes the transcript and returns results"""99    if not transcript or transcript.strip() == "":100        return "", "No segments created", {}101        102    segments = segment_transcript(transcript, max_length, use_smart_boundaries)103    104    # Create segment statistics105    stats = {106        "total_segments": len(segments),107        "total_characters": len(transcript),108        "average_segment_length": len(transcript) / max(1, len(segments)),109        "segments": [110            {"id": i+1, "characters": len(segment), "estimated_tokens": count_tokens(segment)}111            for i, segment in enumerate(segments)112        ]113    }114    115    # Format the segments for display116    formatted_segments = "\n\n" + "\n\n".join(segments)117    118    return formatted_segments, f"{len(segments)} segments created", stats119 120# Create the Gradio interface121with gr.Blocks(title="Transcript Segmenter") as demo:122    gr.Markdown("# Transcript Segmenter")123    gr.Markdown("""124    This tool segments long transcripts into smaller chunks that can be processed by LLM-based tools.125    It intelligently splits at sentence boundaries to maintain context.126    """)127    128    with gr.Row():129        with gr.Column():130            input_text = gr.Textbox(131                label="Full Transcript", 132                placeholder="Paste your full transcript here...",133                lines=10134            )135            136            with gr.Row():137                segment_length = gr.Slider(138                    label="Maximum segment length (characters)",139                    minimum=500,140                    maximum=3000,141                    value=1500,142                    step=100143                )144                145                smart_boundaries = gr.Checkbox(146                    label="Use smart sentence boundaries", 147                    value=True148                )149            150            segment_btn = gr.Button("Segment Transcript")151        152        with gr.Column():153            output_segments = gr.Textbox(154                label="Segmented Transcript",155                placeholder="Segmented transcript will appear here...",156                lines=15157            )158            segment_count = gr.Textbox(label="Number of Segments")159            segment_stats = gr.JSON(label="Segment Statistics")160    161    segment_btn.click(162        fn=process_transcript,163        inputs=[input_text, segment_length, smart_boundaries],164        outputs=[output_segments, segment_count, segment_stats]165    )166    167    # API documentation section168    gr.Markdown("""169    ## API Usage170    171    This tool can be called programmatically using the Gradio client:172    173    ```python174    from gradio_client import Client175    176    client = Client("https://your-space-name.hf.space")177    segments, count, stats = client.predict(178        "Your full transcript text here",  # Input transcript179        1500,                              # Max segment length180        True,                              # Use smart boundaries181        fn_index=0182    )183    ```184    185    ## Tips for Best Results186    187    - The tool works best when transcripts have proper punctuation188    - Using "smart sentence boundaries" preserves coherent segments189    - For transcripts with poor sentence structure, you may want to disable smart boundaries190    - A segment length of 1500-2000 characters works well for most LLM-based analyzers191    """)192 193# Launch the app - this is the standard way to launch Gradio apps194if __name__ == "__main__":195    demo.launch()