import torchfrom transformers import AutoProcessor, Qwen3_5ForConditionalGeneration model_id = "wflying/Qwen3.5-4B-RL-MATH" model = Qwen3_5ForConditionalGeneration.from_pretrained( model_id, torch_dtype="auto", device_map="auto",)processor = AutoProcessor.from_pretrained(model_id) messages = [ { "role": "user", "content": [ { "type": "text", "text": "Solve the problem and put the final answer in \\boxed{}: What is the sum of the first 20 positive integers?", } ], }] inputs = processor.apply_chat_template( messages, tokenize=True, add_generation_prompt=True, enable_thinking=False, return_dict=True, return_tensors="pt",).to(model.device) with torch.inference_mode(): generated_ids = model.generate(**inputs, max_new_tokens=512) generated_ids = generated_ids[:, inputs.input_ids.shape[1]:]print(processor.batch_decode(generated_ids, skip_special_tokens=True)[0])