From 9abccd63cf638d27693fcc6273e156a4556f3221 Mon Sep 17 00:00:00 2001 From: Soulter <905617992@qq.com> Date: Wed, 29 Jan 2025 23:47:50 +0800 Subject: [PATCH] chore: remove stt.py --- stt.py | 33 --------------------------------- 1 file changed, 33 deletions(-) delete mode 100644 stt.py diff --git a/stt.py b/stt.py deleted file mode 100644 index 9eee69b8a..000000000 --- a/stt.py +++ /dev/null @@ -1,33 +0,0 @@ -import torch -from transformers import AutoModelForSpeechSeq2Seq, AutoProcessor, pipeline -from datasets import load_dataset - - -device = "cuda:0" if torch.cuda.is_available() else "cpu" -torch_dtype = torch.float16 if torch.cuda.is_available() else torch.float32 - -model_id = "openai/whisper-large-v3" - -model = AutoModelForSpeechSeq2Seq.from_pretrained( - model_id, torch_dtype=torch_dtype, low_cpu_mem_usage=True -) -model.to(device) - -processor = AutoProcessor.from_pretrained(model_id) - -pipe = pipeline( - "automatic-speech-recognition", - model=model, - tokenizer=processor.tokenizer, - feature_extractor=processor.feature_extractor, - chunk_length_s=30, - batch_size=16, # batch size for inference - set based on your device - torch_dtype=torch_dtype, - device=device, -) - -dataset = load_dataset("distil-whisper/librispeech_long", "clean", split="validation") -sample = dataset[0]["audio"] - -result = pipe(sample) -print(result["text"])