Akjava/Qwen2.5-0.5B-Rag-Thinking-Flan-T5
2
1from langchain.docstore.document import Document2from langchain.text_splitter import RecursiveCharacterTextSplitter3from langchain_community.retrievers import BM25Retriever4import re5# Importing required libraries6import warnings7warnings.filterwarnings("ignore")8import datasets9import os10import json11import subprocess12import sys13import joblib14from llama_cpp import Llama15 16import gradio as gr17from huggingface_hub import hf_hub_download18from typing import List, Tuple,Dict,Optional19from logger import logging20from exception import CustomExceptionHandling21 22cache_file = "docs_processed.joblib"23if os.path.exists(cache_file):24 docs_processed = joblib.load(cache_file)25 #print("Loaded docs_processed from cache.")26else:27 knowledge_base = datasets.load_dataset("m-ric/huggingface_doc", split="train")28 source_docs = [29 Document(page_content=doc["text"], metadata={"source": doc["source"].split("/")[1]}) for doc in knowledge_base30 ]31 32 text_splitter = RecursiveCharacterTextSplitter(33 chunk_size=1000,34 chunk_overlap=50,35 add_start_index=True,36 strip_whitespace=True,37 separators=["\n\n", "\n", ".", " ", ""],38 )39 docs_processed = text_splitter.split_documents(source_docs)40 joblib.dump(docs_processed, cache_file)41 print("Created and saved docs_processed to cache.")42 43class RetrieverTool():44 name = "retriever"45 description = "Uses semantic search to retrieve the parts of documentation that could be most relevant to answer your query."46 inputs = {47 "query": {48 "type": "string",49 "description": "The query to perform. This should be semantically close to your target documents. Use the affirmative form rather than a question.",50 }51 }52 output_type = "string"53 54 def __init__(self, docs, **kwargs):55 #super().__init__(**kwargs)56 57 self.retriever = BM25Retriever.from_documents(58 docs,59 k=7, 60 )61 62 def __call__(self, query: str) -> str:63 assert isinstance(query, str), "Your search query must be a string"64 65 docs = self.retriever.invoke(66 query,67 )68 return "\nRetrieved documents:\n" + "".join(69 [70 f"\n\n===== Document {str(i)} =====\n" + str(doc.page_content)71 for i, doc in enumerate(docs)72 ]73 )74 75 76 77retriever_tool = RetrieverTool(docs_processed)78# Download gguf model files79huggingface_token = os.getenv("HUGGINGFACE_TOKEN")80 81hf_hub_download(82 repo_id="mradermacher/Qwen2.5-0.5B-Rag-Thinking-i1-GGUF",83 filename="Qwen2.5-0.5B-Rag-Thinking.i1-Q6_K.gguf",84 local_dir="./models",85)86 87t5_size="base"88hf_hub_download(89 repo_id=f"Felladrin/gguf-flan-t5-{t5_size}",90 filename=f"flan-t5-{t5_size}.Q8_0.gguf",91 local_dir="./models",92)93 94 95 96 97 98 99 100query_system = """101You are a query rewriter. Your task is to convert a user's question into a concise search query suitable for information retrieval.102The goal is to identify the most important keywords for a search engine.103 104Here are some examples:105 106User Question: What is transformer?107Search Query: transformer108 109User Question: How does a transformer model work in natural language processing?110Search Query: transformer model natural language processing111 112User Question: What are the advantages of using transformers over recurrent neural networks?113Search Query: transformer vs recurrent neural network advantages114 115User Question: Explain the attention mechanism in transformers.116Search Query: transformer attention mechanism117 118User Question: What are the different types of transformer architectures?119Search Query: transformer architectures120 121User Question: What is the history of the transformer model?122Search Query: transformer model history123"""124 125# remove strange char like *,/126def clean_text(text):127 cleaned = re.sub(r'[^\x00-\x7F]+', '', text) # Remove non-ASCII chars128 cleaned = re.sub(r'[^a-zA-Z0-9_\- ]', '', cleaned) #Then your original rule129 cleaned = cleaned.replace("---","")130 return cleaned131 132def generate_t5(llama,message):#text size must be smaller than ctx(default=512)133 if llama == None:134 raise ValueError("llama not initialized")135 try:136 tokens = llama.tokenize(f"{message}".encode("utf-8"))137 #print(f"text length={len(tokens)}")138 llama.encode(tokens)139 tokens = [llama.decoder_start_token()]140 141 142 outputs =""143 144 iteration = 1145 temperature = 0.5146 top_k = 40147 top_p = 0.95148 repeat_penalty = 1.2149 150 for i in range(iteration):151 for token in llama.generate(tokens, top_k=top_k, top_p=top_p, temp=temperature, repeat_penalty=repeat_penalty):152 outputs+= llama.detokenize([token]).decode()153 if token == llama.token_eos():154 break155 return outputs156 except Exception as e:157 raise CustomExceptionHandling(e, sys) from e158 return None159 160 161llama = None162def to_query(question):163 system = """164You are a query rewriter. Your task is to convert a user's question into a concise search query suitable for information retrieval.165The goal is to identify the most important keywords for a search engine.166 167Here are some examples:168User Question: What is transformer?169Search Query: transformer170User Question: How does a transformer model work in natural language processing?171Search Query: transformer model natural language processing172User Question: What are the advantages of using transformers over recurrent neural networks?173Search Query: transformer vs recurrent neural network advantages174User Question: Explain the attention mechanism in transformers.175Search Query: transformer attention mechanism176User Question: What are the different types of transformer architectures?177Search Query: transformer architectures178User Question: What is the history of the transformer model?179Search Query: transformer model history180---181Now, rewrite the following question:182User Question: %s183Search Query:184"""% question185 message = system186 try:187 global llama188 if llama == None:189 model_id = f"flan-t5-{t5_size}.Q8_0.gguf"190 llama = Llama(f"models/{model_id}",flash_attn=False,verbose=False,191 n_gpu_layers=0,192 n_threads=2,193 n_threads_batch=2194 )195 query = generate_t5(llama,message)196 return clean_text(query)197 except Exception as e:198 # Custom exception handling199 raise CustomExceptionHandling(e, sys) from e200 return None201 202 203qwen_prompt = """<|im_start|>system204You answer questions from the user, always using the context provided as a basis.205Write down your reasoning for answering the question, between the <think> and </think> tags.<|im_end|>206<|im_start|>user207Context:208%s209Question:210%s<|im_end|>211<|im_start|>assistant212<think>"""213 214def answer(document:str,question:str,model:str="Qwen2.5-0.5B-Rag-Thinking.i1-Q6_K.gguf")->str:215 global llm216 global llm_model217 global provider218 llm = Llama(219 model_path=f"models/{model}",220 flash_attn=False,221 n_gpu_layers=0,222 n_batch=1024,223 n_ctx=2048*4,224 n_threads=2,225 n_threads_batch=2,226 verbose=False227 )228 llm_model = model229 230def respond(231 message: str,232 history: List[Tuple[str, str]],233 model: str,234 system_message: str,235 max_tokens: int,236 temperature: float,237 top_p: float,238 top_k: int,239 repeat_penalty: float,240):241 """242 Respond to a message using the Gemma3 model via Llama.cpp.243 Args:244 - message (str): The message to respond to.245 - history (List[Tuple[str, str]]): The chat history.246 - model (str): The model to use.247 - system_message (str): The system message to use.248 - max_tokens (int): The maximum number of tokens to generate.249 - temperature (float): The temperature of the model.250 - top_p (float): The top-p of the model.251 - top_k (int): The top-k of the model.252 - repeat_penalty (float): The repetition penalty of the model.253 Returns:254 str: The response to the message.255 """256 if model is None:#257 return258 259 query = to_query(message)260 document = retriever_tool(query=query)261 #print(document)262 answer(document,message)263 response = ""264 #do direct in here265 for chunk in llm(system_message%(document,message),max_tokens=max_tokens,stream=True,top_k=top_k, top_p=top_p, temperature=temperature, repeat_penalty=repeat_penalty):266 text = chunk['choices'][0]['text']267 response += text268 yield response269 270 271# Create a chat interface272# Set the title and description273title = "llama.cpp Qwen2.5-0.5B-Rag-Thinking-Flan-T5"274description = """275- I use forked [llama-cpp-python](https://github.com/fairydreaming/llama-cpp-python/tree/t5) which support T5 on server and it's doesn't support new models(like gemma3)276- Search query generation(query reformulation) Tasks - I use flan-t5-base (large make better result,but too large for just this task)277- Qwen2.5-0.5B as good as small-size.278- anyway google T5 series on CPU is amazing279## Huggingface Free CPU Limitations280- When duplicating a space, the build process can occasionally become stuck, requiring a manual restart to finish.281- Spaces may unexpectedly stop functioning or even be deleted, leading to the need to rework them. Refer to [issue](https://github.com/huggingface/hub-docs/issues/1633) for more information.282"""283 284demo = gr.ChatInterface(285 respond,286 examples=[["What is the Diffuser?"], ["Tell me About Huggingface."], ["How to upload dataset?"]],287 additional_inputs_accordion=gr.Accordion(288 label="⚙️ Parameters", open=False, render=False289 ),290 additional_inputs=[291 gr.Dropdown(292 choices=[293 294 "Qwen2.5-0.5B-Rag-Thinking.i1-Q6_K.gguf",295 ],296 value="Qwen2.5-0.5B-Rag-Thinking.i1-Q6_K.gguf",297 label="Model",298 info="Select the AI model to use for chat",visible=False299 ),300 gr.Textbox(301 value=qwen_prompt,302 label="System Prompt",303 info="Define the AI assistant's personality and behavior",304 lines=2,visible=True305 ),306 gr.Slider(307 minimum=1024,308 maximum=8192,309 value=2048,310 step=1,311 label="Max Tokens",312 info="Maximum length of response (higher = longer replies)",313 ),314 gr.Slider(315 minimum=0.1,316 maximum=2.0,317 value=0.7,318 step=0.1,319 label="Temperature",320 info="Creativity level (higher = more creative, lower = more focused)",321 ),322 gr.Slider(323 minimum=0.1,324 maximum=1.0,325 value=0.95,326 step=0.05,327 label="Top-p",328 info="Nucleus sampling threshold",329 ),330 gr.Slider(331 minimum=1,332 maximum=100,333 value=40,334 step=1,335 label="Top-k",336 info="Limit vocabulary choices to top K tokens",337 ),338 gr.Slider(339 minimum=1.0,340 maximum=2.0,341 value=1.1,342 step=0.1,343 label="Repetition Penalty",344 info="Penalize repeated words (higher = less repetition)",345 ),346 ],347 theme="Ocean",348 submit_btn="Send",349 stop_btn="Stop",350 title=title,351 description=description,352 chatbot=gr.Chatbot(scale=1, show_copy_button=True),353 flagging_mode="never",354)355 356 357# Launch the chat interface358if __name__ == "__main__":359 demo.launch(debug=False)360 