Image-to-Text
Transformers
Safetensors
Japanese
English
sarashina2_vision
text-generation
multimodal
ocr
document-understanding
vision-language
custom_code
Instructions to use sbintuitions/sarashina2.2-ocr with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use sbintuitions/sarashina2.2-ocr with Transformers:
# Use a pipeline as a high-level helper # Warning: Pipeline type "image-to-text" is no longer supported in transformers v5. # You must load the model directly (see below) or downgrade to v4.x with: # 'pip install "transformers<5.0.0' from transformers import pipeline pipe = pipeline("image-to-text", model="sbintuitions/sarashina2.2-ocr", trust_remote_code=True)# Load model directly from transformers import AutoModelForCausalLM model = AutoModelForCausalLM.from_pretrained("sbintuitions/sarashina2.2-ocr", trust_remote_code=True, device_map="auto") - Notebooks
- Google Colab
- Kaggle
| { | |
| "auto_map": { | |
| "AutoProcessor": "processing_sarashina2_vision.Sarashina2VisionProcessor" | |
| }, | |
| "crop_size": null, | |
| "data_format": "channels_first", | |
| "default_to_square": true, | |
| "device": null, | |
| "do_center_crop": null, | |
| "do_convert_rgb": null, | |
| "do_normalize": true, | |
| "do_rescale": true, | |
| "do_resize": true, | |
| "do_sample_frames": true, | |
| "fps": 2, | |
| "fps_max_frames": 64, | |
| "fps_min_frames": 2, | |
| "image_factor": 28, | |
| "image_mean": [ | |
| 0.5, | |
| 0.5, | |
| 0.5 | |
| ], | |
| "image_processor_type": "SiglipImageProcessor", | |
| "image_std": [ | |
| 0.5, | |
| 0.5, | |
| 0.5 | |
| ], | |
| "input_data_format": null, | |
| "max_pixels": 2458624, | |
| "merge_size": 2, | |
| "num_frames": null, | |
| "pad_size": null, | |
| "patch_size": 14, | |
| "processor_class": "Sarashina2VisionProcessor", | |
| "resample": 2, | |
| "rescale_factor": 0.00392156862745098, | |
| "return_metadata": false, | |
| "size": { | |
| "height": 384, | |
| "width": 384 | |
| }, | |
| "temporal_patch_size": 2, | |
| "total_pixels": 2458624, | |
| "video_max_token_num": 768, | |
| "video_metadata": null, | |
| "video_min_token_num": 128, | |
| "video_processor_type": "Sarashina2VisionVideoProcessor" | |
| } | |