Team Ai
Apppublic

Omid-sar/YouTube_Transcript_Analyzer

sourceHugging Faceopenrailupdated 3y agoView on Hugging Face
0likes
app.py70 linesDownload Raw Back to root
1# Import the necessary packages2import os3from langchain.embeddings import OpenAIEmbeddings4from langchain.document_loaders import YoutubeLoader5from langchain.text_splitter import RecursiveCharacterTextSplitter6from langchain.vectorstores import Chroma7from langchain.llms import OpenAI8from langchain.prompts import PromptTemplate9from langchain.chains import LLMChain10import textwrap11import streamlit as st12 13 14 15 16# Load the OpenAI Embeddings, LLM , PromptTemplate and LLMChain17embeddings = OpenAIEmbeddings()18llm = OpenAI(temperature=0)19# Define the template for the prompt20template = """You can provide answers about YouTube videos using their transcripts.21 22    For the question: {question}23    Please refer to the video transcript: {docs_page_content}24 25    Rely solely on the transcript's factual data to respond.26 27    If the information isn't sufficient, simply state "I don't know".28 29    Ensure your answers are comprehensive and in-depth.30    """31prompt = PromptTemplate(32    input_variables=["question", "docs_page_content"],33    template=template,34)35chain = LLMChain(llm=llm, prompt=prompt)36 37 38# Setup streamlit39st.title("YouTube Video Transcript Analyzer")40# *** YOUR VIDEO URL and QUESTION ***41video_url = st.text_input("Enter the YouTube video URL:")42question = st.text_input("Enter your question about the video:")43# add submit button44# submit = st.button("Submit")45#46if video_url and question:47    # load the video transcript48    loader = YoutubeLoader.from_youtube_url(video_url, add_video_info=True)49    # show the video title and author50    info = loader._get_video_info()51    st.write("**Title:**", info["title"])52    st.write("**Author:**", info["author"])53    # Split the transcript into chunks with 1500 characters and 150 characters overlap54    transcript = loader.load()55    text_splitter = RecursiveCharacterTextSplitter(chunk_size=1500, chunk_overlap=150)56    docs = text_splitter.split_documents(transcript)57    # docs[0].page_content58    # Create the vector database which will be used to search for similar sentences59    vectordb = Chroma.from_documents(60        documents=docs, embedding=embeddings, persist_directory="./chroma_db"61    )62 63    # Search for the most similar sentences to the question and concatenate top 3 vectors64    docs = vectordb.similarity_search(query=question, k=3)65    docs_page_content = " ".join([doc.page_content for doc in docs])66    # docs[0].page_content67    # send the question and the top 3 sentences to the LLMChain and print the response68    response = chain.run(question=question, docs_page_content=docs_page_content)69    st.write(textwrap.fill(response, width=85))70