Update transcribe.py
better time keeping
This commit is contained in:
committed by
GitHub
parent
dfe967bd58
commit
4e1c709f43
@@ -1,5 +1,7 @@
|
|||||||
import whisper
|
import whisper
|
||||||
import glob, os
|
import glob, os
|
||||||
|
#import torch #uncomment if using torch with cuda, below too
|
||||||
|
import datetime
|
||||||
|
|
||||||
def transcribe(path, file_type, model=None, language=None, verbose=False):
|
def transcribe(path, file_type, model=None, language=None, verbose=False):
|
||||||
'''Implementation of OpenAI's whisper model. Downloads model, transcribes audio files in a folder and returns the text files with transcriptions'''
|
'''Implementation of OpenAI's whisper model. Downloads model, transcribes audio files in a folder and returns the text files with transcriptions'''
|
||||||
@@ -11,6 +13,11 @@ def transcribe(path, file_type, model=None, language=None, verbose=False):
|
|||||||
|
|
||||||
glob_file = glob.glob(path+'/*{}'.format(file_type))
|
glob_file = glob.glob(path+'/*{}'.format(file_type))
|
||||||
|
|
||||||
|
#if torch.cuda.is_available():
|
||||||
|
# generator = torch.Generator('cuda').manual_seed(42)
|
||||||
|
#else:
|
||||||
|
# generator = torch.Generator().manual_seed(42)
|
||||||
|
|
||||||
print('Using {} model'.format(model))
|
print('Using {} model'.format(model))
|
||||||
print('File type is {}'.format(file_type))
|
print('File type is {}'.format(file_type))
|
||||||
print('Language is being detected automatically for each file')
|
print('Language is being detected automatically for each file')
|
||||||
@@ -34,15 +41,15 @@ def transcribe(path, file_type, model=None, language=None, verbose=False):
|
|||||||
end=[]
|
end=[]
|
||||||
text=[]
|
text=[]
|
||||||
for i in range(len(result['segments'])):
|
for i in range(len(result['segments'])):
|
||||||
start.append(result['segments'][i]['start'])
|
start.append(str(datetime.timedelta(seconds=(result['segments'][i]['start']))))
|
||||||
end.append(result['segments'][i]['end'])
|
end.append(str(datetime.timedelta(seconds=(result['segments'][i]['end']))))
|
||||||
text.append(result['segments'][i]['text'])
|
text.append(result['segments'][i]['text'])
|
||||||
|
|
||||||
with open("{}/transcriptions/{}.txt".format(path,title), 'w', encoding='utf-8') as file:
|
with open("{}/transcriptions/{}.txt".format(path,title), 'w', encoding='utf-8') as file:
|
||||||
file.write(title)
|
file.write(title)
|
||||||
file.write('\nIn seconds:')
|
file.write('\nIn seconds:')
|
||||||
for i in range(len(result['segments'])):
|
for i in range(len(result['segments'])):
|
||||||
file.writelines('\n[{:.2f} --> {:.2f}]:{}'.format(start[i], end[i], text[i]))
|
file.writelines('\n[{} --> {}]:{}'.format(start[i], end[i], text[i]))
|
||||||
|
|
||||||
print('\nFinished file number {}.\n\n\n'.format(idx+1))
|
print('\nFinished file number {}.\n\n\n'.format(idx+1))
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user