from transformers import AutoModelForCausalLM, AutoTokenizer
tok = AutoTokenizer.from_pretrained("Sarath569/slm-125m-legal-raft")
model = AutoModelForCausalLM.from_pretrained("Sarath569/slm-125m-legal-raft")
system = ("You are a legal and financial assistant. Answer the question using ONLY the "
"provided context. If the answer is not contained in the context, say you do "
"not have enough information to answer.")
context = "The Company reported net revenue of $4.2 billion for fiscal 2023, up 12%..."
question = "What was net revenue for fiscal 2023?"
prompt = (f"<|system|>\n{system}\n<|user|>\n<context>\n{context}\n</context>\n\n"
f"Question: {question}\n<|assistant|>\n")
ids = tok(prompt, add_special_tokens=False, return_tensors="pt").input_ids
out = model.generate(ids, max_new_tokens=120, do_sample=False)
print(tok.decode(out[0][ids.shape[1]:], skip_special_tokens=True))