import re, torchfrom transformers import AutoProcessor, AutoModelForImageTextToText model = AutoModelForImageTextToText.from_pretrained( "MonteXiaofeng/MechVL-4B-RL", dtype=torch.bfloat16, device_map="auto")processor = AutoProcessor.from_pretrained("MonteXiaofeng/MechVL-4B-RL") question = "图纸中标注的零件总长度是多少?"# RL format prompt (matches training): question + <think>/<answer> instructionmessages = [{"role": "user", "content": [ {"type": "image", "url": "path/to/drawing.png"}, {"type": "text", "text": question},]}]inputs = processor.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt",).to(model.device)out = model.generate(**inputs, max_new_tokens=4096, do_sample=True, temperature=0.6, top_p=0.95, top_k=20)text = processor.decode(out[0], skip_special_tokens=True)m = re.search(r"<answer>(.*?)</answer>", text, re.DOTALL)print("answer:", m.group(1).strip() if m else text)