This is
whisper-large-v3 model converted to the
OpenVINO™ IR (Intermediate Representation) format with weights compressed to INT8 by
NNCF .
For more information on quantization, check the
OpenVINO model optimization guide .
from datasets import load_dataset
from transformers import AutoProcessor
from optimum.intel.openvino import OVModelForSpeechSeq2Seq
model_id = "OpenVINO/whisper-large-v3-int8-ov"
tokenizer = AutoProcessor.from_pretrained(model_id)
model = OVModelForSpeechSeq2Seq.from_pretrained(model_id)
dataset = load_dataset("hf-internal-testing/librispeech_asr_dummy", "clean", split="validation", trust_remote_code=True)
sample = dataset[0]
input_features = tokenizer(
sample["audio"]["array"],
sampling_rate=sample["audio"]["sampling_rate"],
return_tensors="pt",
).input_features
outputs = model.generate(input_features)
text = tokenizer.batch_decode(outputs)[0]
print(text)
pip install huggingface_hub "datasets<4" librosa soundfile
pip install -U --pre --extra-index-url https://storage.openvinotoolkit.org/simple/wheels/nightly openvino openvino-tokenizers openvino-genai
import huggingface_hub as hf_hub
model_id = "OpenVINO/whisper-large-v3-int8-ov"
model_path = "whisper-large-v3-int8-ov"
hf_hub.snapshot_download(model_id, local_dir=model_path)
import openvino_genai as ov_genai
import datasets
device = "CPU"
pipe = ov_genai.WhisperPipeline(model_path, device)
dataset = datasets.load_dataset("hf-internal-testing/librispeech_asr_dummy", "clean", split="validation", trust_remote_code=True)
sample = dataset[0]["audio"]["array"]
print(pipe.generate(sample))
More GenAI usage examples can be found in OpenVINO GenAI library
docs and
samples
1a. Deploy model on Windows using
binary package :
1 mkdir C:\models
2 ovms.exe --rest_port 8000 --source_model OpenVINO/whisper-large-v3-int8-ov --model_repository_path C:\models
1b. Deploy model in a Docker container:
1 mkdir -p ${ HOME } /models
2 export GPU_ARGS = $( if ls /dev/dri/render* > /dev/null 2 > &1 ; then echo "--device /dev/dri --group-add $( stat -c '%g' /dev/dri/render* | head -n1 ) " ; fi )
3 docker run ${GPU_ARGS} --rm --user $( id -u ) : $( id -g ) -p 8000 :8000 -v ${ HOME } /models:/models openvino/model_server:latest-gpu --rest_port 8000 --model_repository_path /models --source_model OpenVINO/whisper-large-v3-int8-ov
1 import io
2
3 import soundfile as sf
4 from datasets import Audio , load_dataset
5 from openai import OpenAI
6
7
8 dataset = load_dataset (
9 "hf-internal-testing/librispeech_asr_dummy" ,
10 "clean" ,
11 split = "validation" ,
12 ) . cast_column ( "audio" , Audio ( decode = False ) )
13 audio_bytes = dataset [ 0 ] [ "audio" ] [ "bytes" ]
14
15 data , rate = sf . read ( io . BytesIO ( audio_bytes ) )
16 buffer = io . BytesIO ( )
17 sf . write ( buffer , data , rate , format = "WAV" )
18
19 client = OpenAI ( base_url = "http://localhost:8002/v1" , api_key = "not_used" )
20 for event in client . audio . transcriptions . create (
21 model = "OpenVINO/whisper-large-v3-int8-ov" ,
22 file = ( "sample.wav" , buffer . getvalue ( ) ) ,
23 language = "en" ,
24 stream = True ,
25 ) :
26 if getattr ( event , "type" , None ) == "transcript.text.delta" :
27 print ( event . delta , end = "" , flush = True )
28 elif getattr ( event , "type" , None ) == "transcript.text.done" :
29 print ( )
30 break
Check the original model card for
original model card for limitations.
The original model is distributed under
apache-2.0 license. More details can be found in
original model card .
Intel is committed to respecting human rights and avoiding causing or contributing to adverse impacts on human rights. See
Intel’s Global Human Rights Principles . Intel’s products and software are intended only to be used in applications that do not cause or contribute to adverse impacts on human rights.