BlackBoyInAfricaaa/Transcripts
0
1from flask import Flask, request, render_template, jsonify, send_file2from youtube_transcript_api import YouTubeTranscriptApi3from youtube_transcript_api._errors import TranscriptsDisabled, NoTranscriptAvailable, VideoUnavailable4import io5import json6import zipfile7from urllib.parse import urlparse, parse_qs8import re9 10app = Flask(__name__)11 12def extract_video_id(url):13 """14 Extract YouTube video ID from various URL formats15 """16 # Remove any extra spaces or quotes17 url = url.strip().strip('"').strip("'")18 19 # Regular expression patterns for different YouTube URL formats20 patterns = [21 r'(?:youtube\.com\/watch\?v=)([^&]+)', # Standard watch URL22 r'(?:youtu\.be\/)([^&]+)', # Short URL23 r'(?:youtube\.com\/embed\/)([^&]+)', # Embed URL24 r'(?:youtube\.com\/v\/)([^&]+)', # V URL25 r'(?:youtube\.com\/watch\?.*v=)([^&]+)' # URL with additional parameters26 ]27 28 for pattern in patterns:29 match = re.search(pattern, url)30 if match:31 return match.group(1)32 33 # If no pattern matched, try to parse as a direct video ID34 if re.match(r'^[a-zA-Z0-9_-]{11}$', url):35 return url36 37 return None38 39def get_english_transcript(video_id):40 try:41 # First try to get English transcript directly42 transcript = YouTubeTranscriptApi.get_transcript(video_id, languages=['en'])43 return transcript, 'en', 'manual'44 except:45 try:46 # If direct fetch fails, list available transcripts47 transcript_list = YouTubeTranscriptApi.list_transcripts(video_id)48 49 # Try to find English manual transcript50 for transcript in transcript_list:51 if transcript.language_code == 'en' and not transcript.is_generated:52 return transcript.fetch(), 'en', 'manual'53 54 # Try to find English auto-generated transcript55 for transcript in transcript_list:56 if transcript.language_code == 'en' and transcript.is_generated:57 return transcript.fetch(), 'en', 'auto-generated'58 59 # If no English, try to find any manual transcript60 for transcript in transcript_list:61 if not transcript.is_generated:62 return transcript.fetch(), transcript.language_code, 'manual'63 64 # If all else fails, return the first available transcript65 first_transcript = next(iter(transcript_list))66 return first_transcript.fetch(), first_transcript.language_code, 'auto-generated'67 68 except Exception as e:69 raise e70 71@app.route('/')72def index():73 return render_template('index.html')74 75@app.route('/download', methods=['POST'])76def download_transcripts():77 urls_text = request.form.get('urls', '').strip()78 79 # Split by commas and remove any empty strings80 urls = [url.strip() for url in urls_text.split(',') if url.strip()]81 82 if not urls:83 return jsonify({'error': 'No URLs provided'}), 40084 85 results = []86 video_ids = []87 88 # Validate URLs and extract video IDs89 for url in urls:90 video_id = extract_video_id(url)91 if not video_id:92 results.append({93 'url': url,94 'success': False,95 'error': 'Invalid YouTube URL or unable to extract video ID'96 })97 else:98 video_ids.append(video_id)99 results.append({100 'url': url,101 'video_id': video_id,102 'success': True103 })104 105 # Create a ZIP file in memory106 zip_buffer = io.BytesIO()107 with zipfile.ZipFile(zip_buffer, 'w', zipfile.ZIP_DEFLATED) as zip_file:108 successful_downloads = 0109 110 for result in results:111 if not result['success']:112 continue113 114 try:115 # Get the transcript with preference for English116 transcript, language, transcript_type = get_english_transcript(result['video_id'])117 118 # Add metadata to the transcript119 transcript_data = {120 'video_id': result['video_id'],121 'video_url': result['url'],122 'language': language,123 'transcript_type': transcript_type,124 'segments': transcript125 }126 127 # Convert to JSON128 transcript_json = json.dumps(transcript_data, indent=2, ensure_ascii=False)129 130 # Add to ZIP131 zip_file.writestr(f"{result['video_id']}_{language}.json", transcript_json)132 133 # Update result134 result['language'] = language135 result['transcript_type'] = transcript_type136 result['status'] = 'success'137 successful_downloads += 1138 139 except TranscriptsDisabled:140 result['success'] = False141 result['error'] = 'Transcripts disabled for this video'142 except NoTranscriptAvailable:143 result['success'] = False144 result['error'] = 'No transcript available for this video'145 except VideoUnavailable:146 result['success'] = False147 result['error'] = 'Video is unavailable or private'148 except Exception as e:149 result['success'] = False150 result['error'] = f'Unexpected error: {str(e)}'151 152 # Add an error report if any downloads failed153 if successful_downloads < len(results):154 error_report = "Error Report:\n\n"155 for result in results:156 if not result.get('success', True):157 error_report += f"URL: {result['url']}\nError: {result.get('error', 'Unknown error')}\n\n"158 zip_file.writestr("error_report.txt", error_report)159 160 # Check if any transcripts were successfully downloaded161 if successful_downloads == 0:162 error_details = [{"url": r["url"], "error": r.get("error", "Unknown error")} for r in results if not r.get("success", True)]163 return jsonify({164 'error': 'No transcripts could be downloaded',165 'details': error_details166 }), 400167 168 # Prepare response169 zip_buffer.seek(0)170 171 return send_file(172 zip_buffer,173 as_attachment=True,174 download_name='youtube_transcripts.zip',175 mimetype='application/zip'176 )177 178@app.route('/check', methods=['POST'])179def check_transcripts():180 url = request.form.get('url', '').strip()181 if not url:182 return jsonify({'error': 'No URL provided'}), 400183 184 video_id = extract_video_id(url)185 if not video_id:186 return jsonify({'error': 'Invalid YouTube URL'}), 400187 188 try:189 # List available transcripts190 transcript_list = YouTubeTranscriptApi.list_transcripts(video_id)191 192 available_transcripts = []193 for transcript in transcript_list:194 available_transcripts.append({195 'language': transcript.language,196 'language_code': transcript.language_code,197 'is_generated': transcript.is_generated,198 'translation_languages': [{'language': tl[0], 'language_code': tl[1]} for tl in transcript.translation_languages]199 })200 201 return jsonify({202 'video_id': video_id,203 'available_transcripts': available_transcripts204 })205 except TranscriptsDisabled:206 return jsonify({'error': 'Transcripts are disabled for this video'}), 400207 except NoTranscriptAvailable:208 return jsonify({'error': 'No transcripts available for this video'}), 400209 except VideoUnavailable:210 return jsonify({'error': 'Video is unavailable or private'}), 400211 except Exception as e:212 return jsonify({'error': f'Unexpected error: {str(e)}'}), 400213 214if __name__ == '__main__':215 from waitress import serve216 serve(app, host="0.0.0.0", port=7860)