import torchimport gradio as gr from transformers import AutoTokenizer, AutoModelForCausalLMfrom peft import PeftModel BASE_MODEL = "Qwen/Qwen2.5-0.5B-Instruct"LORA_MODEL = "vishnuamarapu/Full-Fine-Tuning-Qwen-2.5-0.5B-instruct-LORA" tokenizer = AutoTokenizer.from_pretrained(BASE_MODEL) base_model = AutoModelForCausalLM.from_pretrained( BASE_MODEL, torch_dtype=torch.float16 if torch.cuda.is_available() else torch.float32, device_map="auto") model = PeftModel.from_pretrained( base_model, LORA_MODEL) model.eval() SYSTEM_PROMPT = ( """ You are Vishnu's personal AI assistant.Always answer using the information you have learned about Vishnu.Do not invent facts.If you do not know the answer, say you don't know.Answer as Vishnu in first person.""") def generate(message, history): messages = [ { "role": "system", "content": SYSTEM_PROMPT, } ] for user, assistant in history: messages.append({"role": "user", "content": user}) messages.append({"role": "assistant", "content": assistant}) messages.append({"role": "user", "content": message}) text = tokenizer.apply_chat_template( messages, tokenize=False, add_generation_prompt=True, ) inputs = tokenizer( text, return_tensors="pt", ).to(model.device) with torch.no_grad(): outputs = model.generate( **inputs, max_new_tokens=256, temperature=0.7, do_sample=True, top_p=0.9, repetition_penalty=1.1, ) answer = tokenizer.decode( outputs[0][inputs.input_ids.shape[-1]:], skip_special_tokens=True, ) return answer gr.ChatInterface( fn=generate, title="Vishnu Personal AI", description="LoRA Fine-Tuned Qwen2.5-0.5B-Instruct",).launch()