yum-archive/TaSTT-Whisper
High-performance GPGPU inference of OpenAI's Whisper automatic speech recognition (ASR) model
git clone https://git.yummers.dev/yum-archive/TaSTT-Whisper
3ba8e63
master
1using Whisper ; 2 3namespace TranscribeCS 4{ 5static class Program 6{ 7static readonly bool streamAudio = true ; 8 9static int Main ( string [] args ) 10{ 11try 12{ 13CommandLineArgs cla ; 14try 15{ 16cla = new CommandLineArgs ( args ); 17} 18catch ( OperationCanceledException ) 19{ 20return 1 ; 21} 22const eLoggerFlags loggerFlags = eLoggerFlags . UseStandardError | eLoggerFlags . SkipFormatMessage ; 23Library . setLogSink ( eLogLevel . Debug , loggerFlags ); 24 25using iModel model = Library . loadModel ( cla . model ); 26using Context context = model . createContext (); 27cla . apply ( ref context . parameters ); 28// When there're multiple input files, assuming they're independent clips 29context . parameters . setFlag ( eFullParamsFlags . NoContext , true ); 30using iMediaFoundation mf = Library . initMediaFoundation (); 31Transcribe transcribe = new Transcribe ( cla ); 32 33foreach ( string audioFile in cla . fileNames ) 34{ 35if ( streamAudio ) 36{ 37using iAudioReader reader = mf . openAudioFile ( audioFile , cla . diarize ); 38context . runFull ( reader , transcribe , null , cla . prompt ); 39} 40else 41{ 42using iAudioBuffer buffer = mf . loadAudioFile ( audioFile , cla . diarize ); 43context . runFull ( buffer , transcribe , cla . prompt ); 44} 45// When asked to, produce these text files 46if ( cla . output_txt ) 47writeTextFile ( context , audioFile ); 48if ( cla . output_srt ) 49writeSubRip ( context , audioFile , cla ); 50if ( cla . output_vtt ) 51writeWebVTT ( context , audioFile ); 52} 53 54context . timingsPrint (); 55return 0 ; 56} 57catch ( Exception ex ) 58{ 59Console . WriteLine ( ex . Message ); 60return ex . HResult ; 61} 62} 63 64static void writeTextFile ( Context context , string audioPath ) 65{ 66using var stream = File . CreateText ( Path . ChangeExtension ( audioPath , ".txt" ) ); 67foreach ( sSegment seg in context . results (). segments ) 68stream . WriteLine ( seg . text ); 69} 70 71static void writeSubRip ( Context context , string audioPath , CommandLineArgs cliArgs ) 72{ 73using var stream = File . CreateText ( Path . ChangeExtension ( audioPath , ".srt" ) ); 74var segments = context . results ( eResultFlags . Timestamps ). segments ; 75 76for ( int i = 0 ; i < segments . Length ; i ++ ) 77{ 78stream . WriteLine ( i + 1 + cliArgs . offset_n ); 79sSegment seg = segments [ i ]; 80string begin = Transcribe . printTimeWithComma ( seg . time . begin ); 81string end = Transcribe . printTimeWithComma ( seg . time . end ); 82stream . WriteLine ( "{0} --> {1}" , begin , end ); 83stream . WriteLine ( seg . text ); 84stream . WriteLine (); 85} 86} 87 88static void writeWebVTT ( Context context , string audioPath ) 89{ 90using var stream = File . CreateText ( Path . ChangeExtension ( audioPath , ".vtt" ) ); 91stream . WriteLine ( "WEBVTT" ); 92stream . WriteLine (); 93 94foreach ( sSegment seg in context . results ( eResultFlags . Timestamps ). segments ) 95{ 96string begin = Transcribe . printTime ( seg . time . begin ); 97string end = Transcribe . printTime ( seg . time . end ); 98stream . WriteLine ( "{0} --> {1}" , begin , end ); 99stream . WriteLine ( seg . text ); 100stream . WriteLine (); 101} 102} 103} 104}