Team Ai
Apppublic

Akjava/Qwen2.5-0.5B-Rag-Thinking-Flan-T5

sourceHugging Facemitupdated 2y agoView on Hugging Face
2likes
app.py360 linesDownload Raw Back to root
1from langchain.docstore.document import Document2from langchain.text_splitter import RecursiveCharacterTextSplitter3from langchain_community.retrievers import BM25Retriever4import re5# Importing required libraries6import warnings7warnings.filterwarnings("ignore")8import datasets9import os10import json11import subprocess12import sys13import joblib14from llama_cpp import Llama15 16import gradio as gr17from huggingface_hub import hf_hub_download18from typing import List, Tuple,Dict,Optional19from logger import logging20from exception import CustomExceptionHandling21 22cache_file = "docs_processed.joblib"23if os.path.exists(cache_file):24    docs_processed = joblib.load(cache_file)25    #print("Loaded docs_processed from cache.")26else:27    knowledge_base = datasets.load_dataset("m-ric/huggingface_doc", split="train")28    source_docs = [29        Document(page_content=doc["text"], metadata={"source": doc["source"].split("/")[1]}) for doc in knowledge_base30    ]31 32    text_splitter = RecursiveCharacterTextSplitter(33        chunk_size=1000,34        chunk_overlap=50,35        add_start_index=True,36        strip_whitespace=True,37        separators=["\n\n", "\n", ".", " ", ""],38    )39    docs_processed = text_splitter.split_documents(source_docs)40    joblib.dump(docs_processed, cache_file)41    print("Created and saved docs_processed to cache.")42 43class RetrieverTool():44    name = "retriever"45    description = "Uses semantic search to retrieve the parts of documentation that could be most relevant to answer your query."46    inputs = {47        "query": {48            "type": "string",49            "description": "The query to perform. This should be semantically close to your target documents. Use the affirmative form rather than a question.",50        }51    }52    output_type = "string"53 54    def __init__(self, docs, **kwargs):55        #super().__init__(**kwargs)56 57        self.retriever = BM25Retriever.from_documents(58            docs,59            k=7,  60        )61 62    def __call__(self, query: str) -> str:63        assert isinstance(query, str), "Your search query must be a string"64 65        docs = self.retriever.invoke(66            query,67        )68        return "\nRetrieved documents:\n" + "".join(69            [70                f"\n\n===== Document {str(i)} =====\n" + str(doc.page_content)71                for i, doc in enumerate(docs)72            ]73        )74 75 76 77retriever_tool = RetrieverTool(docs_processed)78# Download gguf model files79huggingface_token = os.getenv("HUGGINGFACE_TOKEN")80 81hf_hub_download(82    repo_id="mradermacher/Qwen2.5-0.5B-Rag-Thinking-i1-GGUF",83    filename="Qwen2.5-0.5B-Rag-Thinking.i1-Q6_K.gguf",84    local_dir="./models",85)86 87t5_size="base"88hf_hub_download(89    repo_id=f"Felladrin/gguf-flan-t5-{t5_size}",90    filename=f"flan-t5-{t5_size}.Q8_0.gguf",91    local_dir="./models",92)93 94 95 96 97 98 99 100query_system = """101You are a query rewriter. Your task is to convert a user's question into a concise search query suitable for information retrieval.102The goal is to identify the most important keywords for a search engine.103 104Here are some examples:105 106User Question: What is transformer?107Search Query: transformer108 109User Question: How does a transformer model work in natural language processing?110Search Query: transformer model natural language processing111 112User Question: What are the advantages of using transformers over recurrent neural networks?113Search Query: transformer vs recurrent neural network advantages114 115User Question: Explain the attention mechanism in transformers.116Search Query: transformer attention mechanism117 118User Question: What are the different types of transformer architectures?119Search Query: transformer architectures120 121User Question: What is the history of the transformer model?122Search Query: transformer model history123"""124 125# remove strange char like *,/126def clean_text(text):127    cleaned = re.sub(r'[^\x00-\x7F]+', '', text)  # Remove non-ASCII chars128    cleaned = re.sub(r'[^a-zA-Z0-9_\- ]', '', cleaned) #Then your original rule129    cleaned = cleaned.replace("---","")130    return cleaned131    132def generate_t5(llama,message):#text size must be smaller than ctx(default=512)133    if llama == None:134        raise ValueError("llama not initialized")135    try:136        tokens = llama.tokenize(f"{message}".encode("utf-8"))137        #print(f"text length={len(tokens)}")138        llama.encode(tokens)139        tokens = [llama.decoder_start_token()]140        141        142        outputs =""143        144        iteration = 1145        temperature = 0.5146        top_k = 40147        top_p = 0.95148        repeat_penalty = 1.2149        150        for i in range(iteration):151            for token in llama.generate(tokens, top_k=top_k, top_p=top_p, temp=temperature, repeat_penalty=repeat_penalty):152                outputs+= llama.detokenize([token]).decode()153                if token == llama.token_eos():154                    break155        return outputs156    except Exception as e:157        raise CustomExceptionHandling(e, sys) from e158    return None159 160 161llama = None162def to_query(question):163    system = """164You are a query rewriter. Your task is to convert a user's question into a concise search query suitable for information retrieval.165The goal is to identify the most important keywords for a search engine.166 167Here are some examples:168User Question: What is transformer?169Search Query: transformer170User Question: How does a transformer model work in natural language processing?171Search Query: transformer model natural language processing172User Question: What are the advantages of using transformers over recurrent neural networks?173Search Query: transformer vs recurrent neural network advantages174User Question: Explain the attention mechanism in transformers.175Search Query: transformer attention mechanism176User Question: What are the different types of transformer architectures?177Search Query: transformer architectures178User Question: What is the history of the transformer model?179Search Query: transformer model history180---181Now, rewrite the following question:182User Question: %s183Search Query:184"""% question185    message = system186    try:187        global llama188        if llama == None:189            model_id = f"flan-t5-{t5_size}.Q8_0.gguf"190            llama = Llama(f"models/{model_id}",flash_attn=False,verbose=False,191                        n_gpu_layers=0,192                        n_threads=2,193                        n_threads_batch=2194                        )195        query = generate_t5(llama,message)196        return clean_text(query)197    except Exception as e:198        # Custom exception handling199        raise CustomExceptionHandling(e, sys) from e200    return None201 202 203qwen_prompt = """<|im_start|>system204You answer questions from the user, always using the context provided as a basis.205Write down your reasoning for answering the question, between the <think> and </think> tags.<|im_end|>206<|im_start|>user207Context:208%s209Question:210%s<|im_end|>211<|im_start|>assistant212<think>"""213 214def answer(document:str,question:str,model:str="Qwen2.5-0.5B-Rag-Thinking.i1-Q6_K.gguf")->str:215    global llm216    global llm_model217    global provider218    llm = Llama(219                model_path=f"models/{model}",220                flash_attn=False,221                n_gpu_layers=0,222                n_batch=1024,223                n_ctx=2048*4,224                n_threads=2,225                n_threads_batch=2,226                verbose=False227            )228    llm_model = model229 230def respond(231    message: str,232    history: List[Tuple[str, str]],233    model: str,234    system_message: str,235    max_tokens: int,236    temperature: float,237    top_p: float,238    top_k: int,239    repeat_penalty: float,240):241    """242    Respond to a message using the Gemma3 model via Llama.cpp.243    Args:244        - message (str): The message to respond to.245        - history (List[Tuple[str, str]]): The chat history.246        - model (str): The model to use.247        - system_message (str): The system message to use.248        - max_tokens (int): The maximum number of tokens to generate.249        - temperature (float): The temperature of the model.250        - top_p (float): The top-p of the model.251        - top_k (int): The top-k of the model.252        - repeat_penalty (float): The repetition penalty of the model.253    Returns:254        str: The response to the message.255    """256    if model is None:#257        return258 259    query =  to_query(message)260    document = retriever_tool(query=query)261    #print(document)262    answer(document,message)263    response = ""264    #do direct in here265    for chunk in  llm(system_message%(document,message),max_tokens=max_tokens,stream=True,top_k=top_k, top_p=top_p, temperature=temperature, repeat_penalty=repeat_penalty):266        text = chunk['choices'][0]['text']267        response += text268        yield response269 270 271# Create a chat interface272# Set the title and description273title = "llama.cpp Qwen2.5-0.5B-Rag-Thinking-Flan-T5"274description = """275- I use forked [llama-cpp-python](https://github.com/fairydreaming/llama-cpp-python/tree/t5) which support T5 on server and it's doesn't support new models(like gemma3)276- Search query generation(query reformulation) Tasks - I use flan-t5-base (large make better result,but too large for just this task)277- Qwen2.5-0.5B as good as small-size.278- anyway google T5 series on CPU is amazing279## Huggingface Free CPU Limitations280- When duplicating a space, the build process can occasionally become stuck, requiring a manual restart to finish.281- Spaces may unexpectedly stop functioning or even be deleted, leading to the need to rework them. Refer to [issue](https://github.com/huggingface/hub-docs/issues/1633) for more information.282"""283 284demo = gr.ChatInterface(285    respond,286    examples=[["What is the Diffuser?"], ["Tell me About Huggingface."], ["How to upload dataset?"]],287    additional_inputs_accordion=gr.Accordion(288        label="⚙️ Parameters", open=False, render=False289    ),290    additional_inputs=[291        gr.Dropdown(292            choices=[293                294                "Qwen2.5-0.5B-Rag-Thinking.i1-Q6_K.gguf",295            ],296            value="Qwen2.5-0.5B-Rag-Thinking.i1-Q6_K.gguf",297            label="Model",298            info="Select the AI model to use for chat",visible=False299        ),300        gr.Textbox(301            value=qwen_prompt,302            label="System Prompt",303            info="Define the AI assistant's personality and behavior",304            lines=2,visible=True305        ),306        gr.Slider(307            minimum=1024,308            maximum=8192,309            value=2048,310            step=1,311            label="Max Tokens",312            info="Maximum length of response (higher = longer replies)",313        ),314        gr.Slider(315            minimum=0.1,316            maximum=2.0,317            value=0.7,318            step=0.1,319            label="Temperature",320            info="Creativity level (higher = more creative, lower = more focused)",321        ),322        gr.Slider(323            minimum=0.1,324            maximum=1.0,325            value=0.95,326            step=0.05,327            label="Top-p",328            info="Nucleus sampling threshold",329        ),330        gr.Slider(331            minimum=1,332            maximum=100,333            value=40,334            step=1,335            label="Top-k",336            info="Limit vocabulary choices to top K tokens",337        ),338        gr.Slider(339            minimum=1.0,340            maximum=2.0,341            value=1.1,342            step=0.1,343            label="Repetition Penalty",344            info="Penalize repeated words (higher = less repetition)",345        ),346    ],347    theme="Ocean",348    submit_btn="Send",349    stop_btn="Stop",350    title=title,351    description=description,352    chatbot=gr.Chatbot(scale=1, show_copy_button=True),353    flagging_mode="never",354)355 356 357# Launch the chat interface358if __name__ == "__main__":359    demo.launch(debug=False)360