import torchfrom peft import PeftModelfrom transformers import AutoModelForCausalLM, AutoTokenizer, BitsAndBytesConfig BASE, REVISION = "Qwen/Qwen3.5-4B", "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a" tokenizer = AutoTokenizer.from_pretrained(BASE, revision=REVISION)model = AutoModelForCausalLM.from_pretrained( BASE, revision=REVISION, device_map="auto", dtype=torch.bfloat16, # 8 GB GPU: 4-bit base. Remove quantization_config for bf16 on a GPU with >= 12 GB. quantization_config=BitsAndBytesConfig(load_in_4bit=True, bnb_4bit_quant_type="nf4", bnb_4bit_compute_dtype=torch.bfloat16, bnb_4bit_use_double_quant=True),)model = PeftModel.from_pretrained(model, "corners-ai/CoCo-Decision-4B-Ko", revision="v1.0.0").eval() def decide(state: str, question: str, options: dict[str, str]) -> dict[str, float]: """Return a probability for each option key.""" letters = [chr(ord("A") + i) for i in range(len(options))] lines = ["State:", state, "", f"Question: {question}", "Options:"] lines += [f"{letter}. {key}: {text}" for letter, (key, text) in zip(letters, options.items())] lines.append("Answer with the letter only.") prompt = tokenizer.apply_chat_template( [{"role": "user", "content": "\n".join(lines)}], tokenize=False, add_generation_prompt=True ) + "Answer:" ids = tokenizer(prompt, return_tensors="pt", add_special_tokens=False).to(model.device) with torch.no_grad(): logits = model(**ids).logits[0, -1] label_ids = [tokenizer.encode(" " + letter, add_special_tokens=False)[0] for letter in letters] probs = torch.softmax(logits[label_ids].float(), dim=-1).tolist() return dict(zip(options, probs)) print(decide( "Customer: I was charged twice for my order last week.", "Which team should handle this ticket?", {"billing": "payment, refund or invoice problems", "shipping": "delivery status or delays", "account": "login or profile problems"},))# {'billing': 0.986, 'shipping': 0.008, 'account': 0.006} (4-bit, RTX 4060) print(decide( "Refunds are available within 14 days of delivery. The order was delivered 20 days ago.", "Is the customer eligible for a refund?", {"yes": "The statement is true.", "no": "The statement is false."},))# {'yes': 0.064, 'no': 0.936}