import torchfrom transformers import Qwen3VLForConditionalGeneration, AutoProcessor model_id = "XinNUS/CycleGRPO-4B"model = Qwen3VLForConditionalGeneration.from_pretrained( model_id, dtype="auto", device_map="auto").eval()processor = AutoProcessor.from_pretrained(model_id) messages = [{ "role": "user", "content": [ {"type": "image", "image": "figs/totoro.jpg"}, {"type": "text", "text": "Describe the image with interleaved segmentation " "masks for the corresponding parts of the answer."}, ],}]inputs = processor.apply_chat_template( messages, tokenize=True, add_generation_prompt=True, return_dict=True, return_tensors="pt",).to(model.device) out = model.generate(**inputs, max_new_tokens=512, do_sample=False)text = processor.batch_decode(out[:, inputs["input_ids"].shape[1]:], skip_special_tokens=True)[0]print(text) # answer text interleaved with <|mt_start|><|mt_XXXX|><|mt_YYYY|><|mt_end|> mask tokens