from transformers import LlamaForCausalLM, AutoTokenizerimport torch model_name = "AhiskaAI/AhiskaAI-65m-IT-v0.1" # Load the custom-built architecture and vocabularymodel = LlamaForCausalLM.from_pretrained(model_name).to("cuda" if torch.cuda.is_available() else "cpu")tokenizer = AutoTokenizer.from_pretrained(model_name) def ask_ahiska_it(instruction): # Strict Alpaca Template prompt = f"<|im_start|>user\n{user_input}<|im_end|>\n<|im_start|>assistant\n" inputs = tokenizer(prompt, return_tensors="pt").to(model.device) with torch.no_grad(): outputs = model.generate( **inputs, max_length=250, do_sample=True, top_k=40, top_p=0.92, temperature=0.55, # Low temp keeps the 65m nodes highly focused repetition_penalty=1.18 ) response = tokenizer.decode(outputs[0], skip_special_tokens=True) return response.split("### Response:\n")[-1].strip() # Run a test inferenceprint(ask_ahiska_it("Sağlıklı yaşamak için 3 ipucu ver"))