vidore/vidore_v3_computer_science
Viewer • Updated • 8.95k • 3.4k • 6
How to use vanishingradient/qwen-docs-finetuned with Transformers:
# Use a pipeline as a high-level helper
from transformers import pipeline
pipe = pipeline("image-text-to-text", model="vanishingradient/qwen-docs-finetuned")
messages = [
{
"role": "user",
"content": [
{"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"},
{"type": "text", "text": "What animal is on the candy?"}
]
},
]
pipe(text=messages) # Load model directly
from transformers import AutoProcessor, AutoModelForMultimodalLM
processor = AutoProcessor.from_pretrained("vanishingradient/qwen-docs-finetuned")
model = AutoModelForMultimodalLM.from_pretrained("vanishingradient/qwen-docs-finetuned", device_map="auto")
messages = [
{
"role": "user",
"content": [
{"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"},
{"type": "text", "text": "What animal is on the candy?"}
]
},
]
inputs = processor.apply_chat_template(
messages,
add_generation_prompt=True,
tokenize=True,
return_dict=True,
return_tensors="pt",
).to(model.device)
outputs = model.generate(**inputs, max_new_tokens=40)
print(processor.decode(outputs[0][inputs["input_ids"].shape[-1]:]))How to use vanishingradient/qwen-docs-finetuned with vLLM:
# Install vLLM from pip:
pip install vllm
# Start the vLLM server:
vllm serve "vanishingradient/qwen-docs-finetuned"
# Call the server using curl (OpenAI-compatible API):
curl -X POST "http://localhost:8000/v1/chat/completions" \
-H "Content-Type: application/json" \
--data '{
"model": "vanishingradient/qwen-docs-finetuned",
"messages": [
{
"role": "user",
"content": [
{
"type": "text",
"text": "Describe this image in one sentence."
},
{
"type": "image_url",
"image_url": {
"url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg"
}
}
]
}
]
}'docker model run hf.co/vanishingradient/qwen-docs-finetuned
How to use vanishingradient/qwen-docs-finetuned with SGLang:
# Install SGLang from pip:
pip install sglang
# Start the SGLang server:
python3 -m sglang.launch_server \
--model-path "vanishingradient/qwen-docs-finetuned" \
--host 0.0.0.0 \
--port 30000
# Call the server using curl (OpenAI-compatible API):
curl -X POST "http://localhost:30000/v1/chat/completions" \
-H "Content-Type: application/json" \
--data '{
"model": "vanishingradient/qwen-docs-finetuned",
"messages": [
{
"role": "user",
"content": [
{
"type": "text",
"text": "Describe this image in one sentence."
},
{
"type": "image_url",
"image_url": {
"url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg"
}
}
]
}
]
}'docker run --gpus all \
--shm-size 32g \
-p 30000:30000 \
-v ~/.cache/huggingface:/root/.cache/huggingface \
--env "HF_TOKEN=<secret>" \
--ipc=host \
lmsysorg/sglang:latest \
python3 -m sglang.launch_server \
--model-path "vanishingradient/qwen-docs-finetuned" \
--host 0.0.0.0 \
--port 30000
# Call the server using curl (OpenAI-compatible API):
curl -X POST "http://localhost:30000/v1/chat/completions" \
-H "Content-Type: application/json" \
--data '{
"model": "vanishingradient/qwen-docs-finetuned",
"messages": [
{
"role": "user",
"content": [
{
"type": "text",
"text": "Describe this image in one sentence."
},
{
"type": "image_url",
"image_url": {
"url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg"
}
}
]
}
]
}'How to use vanishingradient/qwen-docs-finetuned with Docker Model Runner:
docker model run hf.co/vanishingradient/qwen-docs-finetuned
Developed by: vanishingradient
License: Apache-2.0
Base model: unsloth/Qwen3-VL-8B-Instruct-unsloth-bnb-4bit
This is a fine-tuned Qwen3-VL-8B Vision-Language model optimized for document understanding and structured markdown generation from images such as scanned pages, PDFs, screenshots, and technical documents.
The model was fine-tuned using Unsloth and Hugging Face TRL, enabling faster training and reduced VRAM usage while maintaining output fidelity.
from transformers import AutoModelForVision2Seq, AutoProcessor, TextStreamer
import torch
from PIL import Image
model_id = "vanishingradient/qwen-docs-finetuned"
# Load model (4-bit, fits on 16GB VRAM)
model = AutoModelForVision2Seq.from_pretrained(
model_id,
torch_dtype=torch.float16,
device_map="auto",
trust_remote_code=True,
load_in_4bit=True,
)
processor = AutoProcessor.from_pretrained(
model_id,
trust_remote_code=True
)
# --------------------------------------------------
# PLACEHOLDER: path to your local image file
# --------------------------------------------------
image = Image.open("/path/to/your/document_image.png")
messages = [
{
"role": "user",
"content": [
{"type": "image"},
{"type": "text", "text": "Convert this image to markdown format."}
]
}
]
text = processor.apply_chat_template(
messages,
tokenize=False,
add_generation_prompt=True
)
inputs = processor(
text=[text],
images=[image],
return_tensors="pt"
).to("cuda")
streamer = TextStreamer(
processor.tokenizer,
skip_prompt=True
)
_ = model.generate(
**inputs,
streamer=streamer,
max_new_tokens=1024,
temperature=0.1,
)
Base model
Qwen/Qwen3-VL-8B-Instruct