Omid-sar/YouTube_Transcript_Analyzer
0
1# Import the necessary packages2import os3from langchain.embeddings import OpenAIEmbeddings4from langchain.document_loaders import YoutubeLoader5from langchain.text_splitter import RecursiveCharacterTextSplitter6from langchain.vectorstores import Chroma7from langchain.llms import OpenAI8from langchain.prompts import PromptTemplate9from langchain.chains import LLMChain10import textwrap11import streamlit as st12 13 14 15 16# Load the OpenAI Embeddings, LLM , PromptTemplate and LLMChain17embeddings = OpenAIEmbeddings()18llm = OpenAI(temperature=0)19# Define the template for the prompt20template = """You can provide answers about YouTube videos using their transcripts.21 22 For the question: {question}23 Please refer to the video transcript: {docs_page_content}24 25 Rely solely on the transcript's factual data to respond.26 27 If the information isn't sufficient, simply state "I don't know".28 29 Ensure your answers are comprehensive and in-depth.30 """31prompt = PromptTemplate(32 input_variables=["question", "docs_page_content"],33 template=template,34)35chain = LLMChain(llm=llm, prompt=prompt)36 37 38# Setup streamlit39st.title("YouTube Video Transcript Analyzer")40# *** YOUR VIDEO URL and QUESTION ***41video_url = st.text_input("Enter the YouTube video URL:")42question = st.text_input("Enter your question about the video:")43# add submit button44# submit = st.button("Submit")45#46if video_url and question:47 # load the video transcript48 loader = YoutubeLoader.from_youtube_url(video_url, add_video_info=True)49 # show the video title and author50 info = loader._get_video_info()51 st.write("**Title:**", info["title"])52 st.write("**Author:**", info["author"])53 # Split the transcript into chunks with 1500 characters and 150 characters overlap54 transcript = loader.load()55 text_splitter = RecursiveCharacterTextSplitter(chunk_size=1500, chunk_overlap=150)56 docs = text_splitter.split_documents(transcript)57 # docs[0].page_content58 # Create the vector database which will be used to search for similar sentences59 vectordb = Chroma.from_documents(60 documents=docs, embedding=embeddings, persist_directory="./chroma_db"61 )62 63 # Search for the most similar sentences to the question and concatenate top 3 vectors64 docs = vectordb.similarity_search(query=question, k=3)65 docs_page_content = " ".join([doc.page_content for doc in docs])66 # docs[0].page_content67 # send the question and the top 3 sentences to the LLMChain and print the response68 response = chain.run(question=question, docs_page_content=docs_page_content)69 st.write(textwrap.fill(response, width=85))70 