File size: 2,275 Bytes
191204a
0b93e69
191204a
5864745
191204a
 
 
 
 
 
 
 
 
 
0b93e69
5864745
 
4e2f8e7
 
191204a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5864745
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
from google.cloud import speech
from google.oauth2 import service_account
from google.api_core.exceptions import InvalidArgument
from streamlit import secrets
from config import settings


helper_path = settings.data_path / "MetaHuman_Voice"
user_path = settings.data_path / "User1_Voice"


def transcribe_file(speech_file):
    """Transcribe the given audio file asynchronously."""

    # Create API client.
    credentials = service_account.Credentials.from_service_account_info(secrets)
    # credentials = service_account.Credentials.from_service_account_file(settings.base_path / "credentials.json")
    client = speech.SpeechClient(credentials=credentials)
    # client = speech.SpeechClient()

    with open(speech_file, "rb") as audio_file:
        content = audio_file.read()

    """
     Note that transcription is limited to a 60 seconds audio file.
     Use a GCS file for audio longer than 1 minute.
    """
    audio = speech.RecognitionAudio(content=content)

    try:
        config = speech.RecognitionConfig(
            encoding=speech.RecognitionConfig.AudioEncoding.LINEAR16,
            language_code="en-US",
            audio_channel_count=2,
        )

        operation = client.long_running_recognize(config=config, audio=audio)
    
    except InvalidArgument as e:
        config = speech.RecognitionConfig(
            encoding=speech.RecognitionConfig.AudioEncoding.LINEAR16,
            language_code="en-US",
        )

        operation = client.long_running_recognize(config=config, audio=audio)


    # print("Waiting for operation to complete...")
    response = operation.result(timeout=90)

    # Each result is for a consecutive portion of the audio. Iterate through
    # them to get the transcripts for the entire audio file.
    for result in response.results:
        # The first alternative is the most likely one for this portion.
        print(u"(transcript) {}".format(result.alternatives[0].transcript))
        # print("Confidence: {}".format(result.alternatives[0].confidence))

    if response.results:
        text = response.results[0].alternatives[0].transcript
    else:
        text = ""

    return text
    

if __name__ == "__main__":

    transcribe_file(helper_path / "Video_010_Would-you-like-to-know-more.wav")