import torchfrom transformers import AutoModelForVision2Seq, AutoProcessorfrom PIL import Image # Load model and processormodel = AutoModelForVision2Seq.from_pretrained( "korea-deep-learning/Qwen2.5-VL-7B-TableRecog", torch_dtype=torch.bfloat16, device_map="auto", trust_remote_code=True) processor = AutoProcessor.from_pretrained( "korea-deep-learning/Qwen2.5-VL-7B-TableRecog", trust_remote_code=True) # Prepare image and textimage = Image.open("table_image.png")text = "Extract the table content and structure from this image." # Process inputsinputs = processor(text=text, images=image, return_tensors="pt").to(model.device) # Generate outputwith torch.no_grad(): outputs = model.generate(**inputs, max_new_tokens=1024) result = processor.decode(outputs[0], skip_special_tokens=True)print(result)