from google.cloud import speech from google.oauth2 import service_account from google.api_core.exceptions import InvalidArgument from streamlit import secrets from config import settings helper_path = settings.data_path / "MetaHuman_Voice" user_path = settings.data_path / "User1_Voice" def transcribe_file(speech_file): """Transcribe the given audio file asynchronously.""" # Create API client. credentials = service_account.Credentials.from_service_account_info(secrets) # credentials = service_account.Credentials.from_service_account_file(settings.base_path / "credentials.json") client = speech.SpeechClient(credentials=credentials) # client = speech.SpeechClient() with open(speech_file, "rb") as audio_file: content = audio_file.read() """ Note that transcription is limited to a 60 seconds audio file. Use a GCS file for audio longer than 1 minute. """ audio = speech.RecognitionAudio(content=content) try: config = speech.RecognitionConfig( encoding=speech.RecognitionConfig.AudioEncoding.LINEAR16, language_code="en-US", audio_channel_count=2, ) operation = client.long_running_recognize(config=config, audio=audio) except InvalidArgument as e: config = speech.RecognitionConfig( encoding=speech.RecognitionConfig.AudioEncoding.LINEAR16, language_code="en-US", ) operation = client.long_running_recognize(config=config, audio=audio) # print("Waiting for operation to complete...") response = operation.result(timeout=90) # Each result is for a consecutive portion of the audio. Iterate through # them to get the transcripts for the entire audio file. for result in response.results: # The first alternative is the most likely one for this portion. print(u"(transcript) {}".format(result.alternatives[0].transcript)) # print("Confidence: {}".format(result.alternatives[0].confidence)) if response.results: text = response.results[0].alternatives[0].transcript else: text = "" return text if __name__ == "__main__": transcribe_file(helper_path / "Video_010_Would-you-like-to-know-more.wav")