abdulshakur/YT-TranscriptSegmenter
0
1import gradio as gr2import nltk3import re4from nltk.tokenize import sent_tokenize5 6# Make sure NLTK data is downloaded correctly7import os8os.environ['NLTK_DATA'] = '/home/user/nltk_data'9nltk.download('punkt', quiet=True, download_dir='/home/user/nltk_data')10 11# Make sure the data is available12try:13 nltk.data.find('tokenizers/punkt')14 print("NLTK punkt tokenizer found successfully!")15except LookupError as e:16 print(f"Error finding punkt tokenizer: {e}")17 # Try a more explicit download18 nltk.download('punkt', download_dir='/home/user/nltk_data')19 print("Attempted explicit download of punkt")20 21def count_tokens(text):22 """23 Estimate the number of tokens in a text.24 This is a rough approximation based on counting words and punctuation.25 """26 # Split on whitespace and keep punctuation as tokens27 try:28 words = re.findall(r'\b\w+\b|[.,!?;:]', text)29 return len(words)30 except Exception as e:31 print(f"Error counting tokens: {e}")32 # Fallback to a simpler method33 return len(text.split())34 35def segment_transcript(transcript, max_segment_length=1500, smart_boundaries=True):36 """37 Segments a transcript into smaller chunks for processing.38 39 Args:40 transcript: The full transcript text41 max_segment_length: Maximum length of each segment in characters42 smart_boundaries: Whether to use sentence boundaries for smarter segmentation43 44 Returns:45 A list of segments46 """47 if not transcript or transcript.strip() == "":48 return []49 50 # Clean up the transcript by normalizing whitespace51 transcript = re.sub(r'\s+', ' ', transcript).strip()52 53 if smart_boundaries:54 try:55 # Use sentence tokenization for smarter segmentation56 sentences = sent_tokenize(transcript)57 print(f"Successfully tokenized transcript into {len(sentences)} sentences")58 except Exception as e:59 print(f"Error during sentence tokenization: {e}")60 print("Falling back to simple segmentation")61 # Fall back to simple segmentation62 smart_boundaries = False63 64 if smart_boundaries:65 segments = []66 current_segment = ""67 current_token_count = 068 estimated_token_limit = max_segment_length # Characters as rough approximation69 70 for sentence in sentences:71 sentence_token_count = count_tokens(sentence)72 73 if current_token_count + sentence_token_count <= estimated_token_limit:74 current_segment += sentence + " "75 current_token_count += sentence_token_count76 else:77 if current_segment:78 segments.append(current_segment.strip())79 current_segment = sentence + " "80 current_token_count = sentence_token_count81 82 if current_segment: # Add the last segment if it exists83 segments.append(current_segment.strip())84 else:85 # Simple character-based chunking without respecting sentence boundaries86 segments = []87 for i in range(0, len(transcript), max_segment_length):88 segments.append(transcript[i:i + max_segment_length])89 90 # Add segment numbers for easy reference91 numbered_segments = [f"Segment {i+1}/{len(segments)}:\n{segment}" 92 for i, segment in enumerate(segments)]93 94 print(f"Created {len(numbered_segments)} segments")95 return numbered_segments96 97def process_transcript(transcript, max_length, use_smart_boundaries):98 """Main function that processes the transcript and returns results"""99 if not transcript or transcript.strip() == "":100 return "", "No segments created", {}101 102 segments = segment_transcript(transcript, max_length, use_smart_boundaries)103 104 # Create segment statistics105 stats = {106 "total_segments": len(segments),107 "total_characters": len(transcript),108 "average_segment_length": len(transcript) / max(1, len(segments)),109 "segments": [110 {"id": i+1, "characters": len(segment), "estimated_tokens": count_tokens(segment)}111 for i, segment in enumerate(segments)112 ]113 }114 115 # Format the segments for display116 formatted_segments = "\n\n" + "\n\n".join(segments)117 118 return formatted_segments, f"{len(segments)} segments created", stats119 120# Create the Gradio interface121with gr.Blocks(title="Transcript Segmenter") as demo:122 gr.Markdown("# Transcript Segmenter")123 gr.Markdown("""124 This tool segments long transcripts into smaller chunks that can be processed by LLM-based tools.125 It intelligently splits at sentence boundaries to maintain context.126 """)127 128 with gr.Row():129 with gr.Column():130 input_text = gr.Textbox(131 label="Full Transcript", 132 placeholder="Paste your full transcript here...",133 lines=10134 )135 136 with gr.Row():137 segment_length = gr.Slider(138 label="Maximum segment length (characters)",139 minimum=500,140 maximum=3000,141 value=1500,142 step=100143 )144 145 smart_boundaries = gr.Checkbox(146 label="Use smart sentence boundaries", 147 value=True148 )149 150 segment_btn = gr.Button("Segment Transcript")151 152 with gr.Column():153 output_segments = gr.Textbox(154 label="Segmented Transcript",155 placeholder="Segmented transcript will appear here...",156 lines=15157 )158 segment_count = gr.Textbox(label="Number of Segments")159 segment_stats = gr.JSON(label="Segment Statistics")160 161 segment_btn.click(162 fn=process_transcript,163 inputs=[input_text, segment_length, smart_boundaries],164 outputs=[output_segments, segment_count, segment_stats]165 )166 167 # API documentation section168 gr.Markdown("""169 ## API Usage170 171 This tool can be called programmatically using the Gradio client:172 173 ```python174 from gradio_client import Client175 176 client = Client("https://your-space-name.hf.space")177 segments, count, stats = client.predict(178 "Your full transcript text here", # Input transcript179 1500, # Max segment length180 True, # Use smart boundaries181 fn_index=0182 )183 ```184 185 ## Tips for Best Results186 187 - The tool works best when transcripts have proper punctuation188 - Using "smart sentence boundaries" preserves coherent segments189 - For transcripts with poor sentence structure, you may want to disable smart boundaries190 - A segment length of 1500-2000 characters works well for most LLM-based analyzers191 """)192 193# Launch the app - this is the standard way to launch Gradio apps194if __name__ == "__main__":195 demo.launch()