Team Ai
Apppublic

BlackBoyInAfricaaa/Transcripts

sourceHugging Facemitupdated 1y agoView on Hugging Face
0likes
app.py216 linesDownload Raw Back to root
1from flask import Flask, request, render_template, jsonify, send_file2from youtube_transcript_api import YouTubeTranscriptApi3from youtube_transcript_api._errors import TranscriptsDisabled, NoTranscriptAvailable, VideoUnavailable4import io5import json6import zipfile7from urllib.parse import urlparse, parse_qs8import re9 10app = Flask(__name__)11 12def extract_video_id(url):13    """14    Extract YouTube video ID from various URL formats15    """16    # Remove any extra spaces or quotes17    url = url.strip().strip('"').strip("'")18    19    # Regular expression patterns for different YouTube URL formats20    patterns = [21        r'(?:youtube\.com\/watch\?v=)([^&]+)',  # Standard watch URL22        r'(?:youtu\.be\/)([^&]+)',              # Short URL23        r'(?:youtube\.com\/embed\/)([^&]+)',    # Embed URL24        r'(?:youtube\.com\/v\/)([^&]+)',        # V URL25        r'(?:youtube\.com\/watch\?.*v=)([^&]+)' # URL with additional parameters26    ]27    28    for pattern in patterns:29        match = re.search(pattern, url)30        if match:31            return match.group(1)32    33    # If no pattern matched, try to parse as a direct video ID34    if re.match(r'^[a-zA-Z0-9_-]{11}$', url):35        return url36    37    return None38 39def get_english_transcript(video_id):40    try:41        # First try to get English transcript directly42        transcript = YouTubeTranscriptApi.get_transcript(video_id, languages=['en'])43        return transcript, 'en', 'manual'44    except:45        try:46            # If direct fetch fails, list available transcripts47            transcript_list = YouTubeTranscriptApi.list_transcripts(video_id)48            49            # Try to find English manual transcript50            for transcript in transcript_list:51                if transcript.language_code == 'en' and not transcript.is_generated:52                    return transcript.fetch(), 'en', 'manual'53            54            # Try to find English auto-generated transcript55            for transcript in transcript_list:56                if transcript.language_code == 'en' and transcript.is_generated:57                    return transcript.fetch(), 'en', 'auto-generated'58            59            # If no English, try to find any manual transcript60            for transcript in transcript_list:61                if not transcript.is_generated:62                    return transcript.fetch(), transcript.language_code, 'manual'63            64            # If all else fails, return the first available transcript65            first_transcript = next(iter(transcript_list))66            return first_transcript.fetch(), first_transcript.language_code, 'auto-generated'67            68        except Exception as e:69            raise e70 71@app.route('/')72def index():73    return render_template('index.html')74 75@app.route('/download', methods=['POST'])76def download_transcripts():77    urls_text = request.form.get('urls', '').strip()78    79    # Split by commas and remove any empty strings80    urls = [url.strip() for url in urls_text.split(',') if url.strip()]81    82    if not urls:83        return jsonify({'error': 'No URLs provided'}), 40084    85    results = []86    video_ids = []87    88    # Validate URLs and extract video IDs89    for url in urls:90        video_id = extract_video_id(url)91        if not video_id:92            results.append({93                'url': url,94                'success': False,95                'error': 'Invalid YouTube URL or unable to extract video ID'96            })97        else:98            video_ids.append(video_id)99            results.append({100                'url': url,101                'video_id': video_id,102                'success': True103            })104    105    # Create a ZIP file in memory106    zip_buffer = io.BytesIO()107    with zipfile.ZipFile(zip_buffer, 'w', zipfile.ZIP_DEFLATED) as zip_file:108        successful_downloads = 0109        110        for result in results:111            if not result['success']:112                continue113                114            try:115                # Get the transcript with preference for English116                transcript, language, transcript_type = get_english_transcript(result['video_id'])117                118                # Add metadata to the transcript119                transcript_data = {120                    'video_id': result['video_id'],121                    'video_url': result['url'],122                    'language': language,123                    'transcript_type': transcript_type,124                    'segments': transcript125                }126                127                # Convert to JSON128                transcript_json = json.dumps(transcript_data, indent=2, ensure_ascii=False)129                130                # Add to ZIP131                zip_file.writestr(f"{result['video_id']}_{language}.json", transcript_json)132                133                # Update result134                result['language'] = language135                result['transcript_type'] = transcript_type136                result['status'] = 'success'137                successful_downloads += 1138                139            except TranscriptsDisabled:140                result['success'] = False141                result['error'] = 'Transcripts disabled for this video'142            except NoTranscriptAvailable:143                result['success'] = False144                result['error'] = 'No transcript available for this video'145            except VideoUnavailable:146                result['success'] = False147                result['error'] = 'Video is unavailable or private'148            except Exception as e:149                result['success'] = False150                result['error'] = f'Unexpected error: {str(e)}'151        152        # Add an error report if any downloads failed153        if successful_downloads < len(results):154            error_report = "Error Report:\n\n"155            for result in results:156                if not result.get('success', True):157                    error_report += f"URL: {result['url']}\nError: {result.get('error', 'Unknown error')}\n\n"158            zip_file.writestr("error_report.txt", error_report)159    160    # Check if any transcripts were successfully downloaded161    if successful_downloads == 0:162        error_details = [{"url": r["url"], "error": r.get("error", "Unknown error")} for r in results if not r.get("success", True)]163        return jsonify({164            'error': 'No transcripts could be downloaded',165            'details': error_details166        }), 400167    168    # Prepare response169    zip_buffer.seek(0)170    171    return send_file(172        zip_buffer,173        as_attachment=True,174        download_name='youtube_transcripts.zip',175        mimetype='application/zip'176    )177 178@app.route('/check', methods=['POST'])179def check_transcripts():180    url = request.form.get('url', '').strip()181    if not url:182        return jsonify({'error': 'No URL provided'}), 400183    184    video_id = extract_video_id(url)185    if not video_id:186        return jsonify({'error': 'Invalid YouTube URL'}), 400187    188    try:189        # List available transcripts190        transcript_list = YouTubeTranscriptApi.list_transcripts(video_id)191        192        available_transcripts = []193        for transcript in transcript_list:194            available_transcripts.append({195                'language': transcript.language,196                'language_code': transcript.language_code,197                'is_generated': transcript.is_generated,198                'translation_languages': [{'language': tl[0], 'language_code': tl[1]} for tl in transcript.translation_languages]199            })200        201        return jsonify({202            'video_id': video_id,203            'available_transcripts': available_transcripts204        })205    except TranscriptsDisabled:206        return jsonify({'error': 'Transcripts are disabled for this video'}), 400207    except NoTranscriptAvailable:208        return jsonify({'error': 'No transcripts available for this video'}), 400209    except VideoUnavailable:210        return jsonify({'error': 'Video is unavailable or private'}), 400211    except Exception as e:212        return jsonify({'error': f'Unexpected error: {str(e)}'}), 400213 214if __name__ == '__main__':215    from waitress import serve216    serve(app, host="0.0.0.0", port=7860)