Views
No views yet

pip install einops torchvision1from transformers import AutoModelForCausalLM, AutoProcessor, GenerationConfig
2from PIL import Image
3import requests
4
5# load the processor
6processor = AutoProcessor.from_pretrained(
7 'allenai/MolmoE-1B-0924',
8 trust_remote_code=True,
9 torch_dtype='auto',
10 device_map='auto'
11)
12
13# load the model
14model = AutoModelForCausalLM.from_pretrained(
15 'allenai/MolmoE-1B-0924',
16 trust_remote_code=True,
17 torch_dtype='auto',
18 device_map='auto'
19)
20
21# process the image and text
22inputs = processor.process(
23 images=[Image.open(requests.get("https://picsum.photos/id/237/536/354", stream=True).raw)],
24 text="Describe this image."
25)
26
27# move inputs to the correct device and make a batch of size 1
28inputs = {k: v.to(model.device).unsqueeze(0) for k, v in inputs.items()}
29
30# generate output; maximum 200 new tokens; stop generation when <|endoftext|> is generated
31output = model.generate_from_batch(
32 inputs,
33 GenerationConfig(max_new_tokens=200, stop_strings="<|endoftext|>"),
34 tokenizer=processor.tokenizer
35)
36
37# only get generated tokens; decode them to text
38generated_tokens = output[0,inputs['input_ids'].size(1):]
39generated_text = processor.tokenizer.decode(generated_tokens, skip_special_tokens=True)
40
41# print the generated text
42print(generated_text)
43
44# >>> This photograph captures a small black puppy, likely a Labrador or a similar breed,
45# sitting attentively on a weathered wooden deck. The deck, composed of three...| Model | Average Score on 11 Academic Benchmarks | Human Preference Elo Rating |
|---|---|---|
| Molmo 72B | 81.2 | 1077 |
| Molmo 7B-D | 77.3 | 1056 |
| Molmo 7B-O | 74.6 | 1051 |
| MolmoE 1B (this model) | 68.6 | 1032 |
| GPT-4o | 78.5 | 1079 |
| GPT-4V | 71.1 | 1041 |
| Gemini 1.5 Pro | 78.3 | 1074 |
| Gemini 1.5 Flash | 75.1 | 1054 |
| Claude 3.5 Sonnet | 76.7 | 1069 |
| Claude 3 Opus | 66.4 | 971 |
| Claude 3 Haiku | 65.3 | 999 |
| Qwen VL2 72B | 79.4 | 1037 |
| Qwen VL2 7B | 73.7 | 1025 |
| Intern VL2 LLAMA 76B | 77.1 | 1018 |
| Intern VL2 8B | 69.4 | 953 |
| Pixtral 12B | 69.5 | 1016 |
| Phi3.5-Vision 4B | 59.7 | 982 |
| PaliGemma 3B | 50.0 | 937 |
| LLAVA OneVision 72B | 76.6 | 1051 |
| LLAVA OneVision 7B | 72.0 | 1024 |
| Cambrian-1 34B | 66.8 | 953 |
| Cambrian-1 8B | 63.4 | 952 |
| xGen - MM - Interleave 4B | 59.5 | 979 |
| LLAVA-1.5 13B | 43.9 | 960 |
| LLAVA-1.5 7B | 40.7 | 951 |
1from PIL import Image
2
3image = Image.open(...)
4
5if image.mode != "RGB":
6 image = image.convert("RGB")1
2# Load the image
3url = "..."
4image = Image.open(requests.get(url, stream=True).raw)
5
6# Convert the image to grayscale to calculate brightness
7gray_image = image.convert('L') # Convert to grayscale
8
9# Calculate the average brightness
10stat = ImageStat.Stat(gray_image)
11average_brightness = stat.mean[0] # Get the average value
12
13# Define background color based on brightness (threshold can be adjusted)
14bg_color = (0, 0, 0) if average_brightness > 127 else (255, 255, 255)
15
16# Create a new image with the same size as the original, filled with the background color
17new_image = Image.new('RGB', image.size, bg_color)
18
19# Paste the original image on top of the background (use image as a mask if needed)
20new_image.paste(image, (0, 0), image if image.mode == 'RGBA' else None)
21
22# Now you can pass the new_image to Molmo
23processor = AutoProcessor.from_pretrained(
24 'allenai/Molmo-7B-D-0924',
25 trust_remote_code=True,
26 torch_dtype='auto',
27 device_map='auto'
28)