Views
No views yet

pip install einops torchvision1from transformers import AutoModelForCausalLM, AutoProcessor, GenerationConfig
2from PIL import Image
3import requests
4import torch
5
6# load the processor
7processor = AutoProcessor.from_pretrained(
8 'allenai/Molmo-72B-0924',
9 trust_remote_code=True,
10 torch_dtype='auto',
11 device_map='auto'
12)
13
14# load the model
15model = AutoModelForCausalLM.from_pretrained(
16 'allenai/Molmo-72B-0924',
17 trust_remote_code=True,
18 torch_dtype='auto',
19 device_map='auto'
20)
21
22# process the image and text
23inputs = processor.process(
24 images=[Image.open(requests.get("https://picsum.photos/id/237/536/354", stream=True).raw)],
25 text="Describe this image."
26)
27
28# move inputs to the correct device and make a batch of size 1
29inputs = {k: v.to(model.device).unsqueeze(0) for k, v in inputs.items()}
30
31# generate output; maximum 200 new tokens; stop generation when <|endoftext|> is generated
32output = model.generate_from_batch(
33 inputs,
34 GenerationConfig(max_new_tokens=200, stop_strings="<|endoftext|>"),
35 tokenizer=processor.tokenizer
36)
37
38# only get generated tokens; decode them to text
39generated_tokens = output[0,inputs['input_ids'].size(1):]
40generated_text = processor.tokenizer.decode(generated_tokens, skip_special_tokens=True)
41
42# print the generated text
43print(generated_text)
44
45# >>> This image features an adorable black Labrador puppy sitting on a wooden deck.
46# The puppy is positioned in the center of the frame, looking up at the camera...1with torch.autocast(device_type="cuda", enabled=True, dtype=torch.bfloat16):
2 output = model.generate_from_batch(
3 inputs,
4 GenerationConfig(max_new_tokens=200, stop_strings="<|endoftext|>"),
5 tokenizer=processor.tokenizer
6 )model.to(dtype=torch.bfloat16)
inputs["images"] = inputs["images"].to(torch.bfloat16)
output = model.generate_from_batch(
inputs,
GenerationConfig(max_new_tokens=200, stop_strings="<|endoftext|>"),
tokenizer=processor.tokenizer
)| Model | Average Score on 11 Academic Benchmarks | Human Preference Elo Rating |
|---|---|---|
| Molmo 72B (this model) | 81.2 | 1077 |
| Molmo 7B-D | 77.3 | 1056 |
| Molmo 7B-O | 74.6 | 1051 |
| MolmoE 1B | 68.6 | 1032 |
| GPT-4o | 78.5 | 1079 |
| GPT-4V | 71.1 | 1041 |
| Gemini 1.5 Pro | 78.3 | 1074 |
| Gemini 1.5 Flash | 75.1 | 1054 |
| Claude 3.5 Sonnet | 76.7 | 1069 |
| Claude 3 Opus | 66.4 | 971 |
| Claude 3 Haiku | 65.3 | 999 |
| Qwen VL2 72B | 79.4 | 1037 |
| Qwen VL2 7B | 73.7 | 1025 |
| Intern VL2 LLAMA 76B | 77.1 | 1018 |
| Intern VL2 8B | 69.4 | 953 |
| Pixtral 12B | 69.5 | 1016 |
| Phi3.5-Vision 4B | 59.7 | 982 |
| PaliGemma 3B | 50.0 | 937 |
| LLAVA OneVision 72B | 76.6 | 1051 |
| LLAVA OneVision 7B | 72.0 | 1024 |
| Cambrian-1 34B | 66.8 | 953 |
| Cambrian-1 8B | 63.4 | 952 |
| xGen - MM - Interleave 4B | 59.5 | 979 |
| LLAVA-1.5 13B | 43.9 | 960 |
| LLAVA-1.5 7B | 40.7 | 951 |
1from PIL import Image
2
3image = Image.open(...)
4
5if image.mode != "RGB":
6 image = image.convert("RGB")1
2# Load the image
3url = "..."
4image = Image.open(requests.get(url, stream=True).raw)
5
6# Convert the image to grayscale to calculate brightness
7gray_image = image.convert('L') # Convert to grayscale
8
9# Calculate the average brightness
10stat = ImageStat.Stat(gray_image)
11average_brightness = stat.mean[0] # Get the average value
12
13# Define background color based on brightness (threshold can be adjusted)
14bg_color = (0, 0, 0) if average_brightness > 127 else (255, 255, 255)
15
16# Create a new image with the same size as the original, filled with the background color
17new_image = Image.new('RGB', image.size, bg_color)
18
19# Paste the original image on top of the background (use image as a mask if needed)
20new_image.paste(image, (0, 0), image if image.mode == 'RGBA' else None)
21
22# Now you can pass the new_image to Molmo
23processor = AutoProcessor.from_pretrained(
24 'allenai/Molmo-7B-D-0924',
25 trust_remote_code=True,
26 torch_dtype='auto',
27 device_map='auto'
28)