File size: 2,715 Bytes
3192caa
 
 
 
 
 
 
 
 
 
 
 
b1d00fe
 
 
 
 
9bfb002
b1d00fe
8bd42ec
b1d00fe
 
b888dbe
b1d00fe
 
4581df8
3fcd7c6
b1d00fe
 
53cbaf1
9bfb002
7214776
3192caa
cbc3ef2
3192caa
 
6efa823
ef99dbf
6efa823
3192caa
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5d5d4d1
 
 
3192caa
 
 
5d5d4d1
cbc3ef2
 
3192caa
 
6efa823
 
3192caa
 
 
ef99dbf
 
 
08c21c5
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
import streamlit as st
from transformers import pipeline

# function part
# img2text
def img2text(url):
    image_to_text_model = pipeline("image-to-text", model="Salesforce/blip-image-captioning-base")
    text = image_to_text_model(url)[0]["generated_text"]
    return text

# text2story
def text2story(text):
    #messages = f"""Generate a vivid story in 100-120 words using the following sentences: "{text}".
    #Structure requirements:
    #1. Only 1 paragraphs and the story mest have beginning, middle, end
    #2. Include at least two named characters
    #3. Use descriptive language
    
    #Story:"""
    #messages = f"""Generate a story of about 100 words using the following sentences: {text}."""
    messages = [
        {"role": "system", "content": "You are Qwen, created by Alibaba Cloud. You are a helpful assistant."},
        {"role": "user", "content": f"""Please help generate a vivid story in 200 words using the following sentences: "{text}".The story must contain:Beginning, middle, end and use descriptive language"""},
    ]
    text_to_story_model= pipeline("text-generation", model="Qwen/Qwen2.5-0.5B-Instruct")
    story_text = text_to_story_model(
        messages,
        max_length=400,  
        min_length=100,
        do_sample=True,
        temperature=0.65
    )[0]['generated_text'][2]['content']
    return story_text
    
# text2audio
def text2audio(story_text):
    text_to_audio_model = pipeline("text-to-speech", model="facebook/mms-tts-eng")
    #audio_data = ""     # to be completed
    audio_data = text_to_audio_model(story_text)
    return audio_data

st.set_page_config(page_title="Your Image to Audio Story",
                   page_icon="🦜")
st.header("Turn Your Image to Audio Story")
uploaded_file = st.file_uploader("Select an Image...")

if uploaded_file is not None:
    print(uploaded_file)
    bytes_data = uploaded_file.getvalue()
    with open(uploaded_file.name, "wb") as file:
        file.write(bytes_data)
    st.image(uploaded_file, caption="Uploaded Image",
             use_column_width=True)

    #Stage 1: Image to Text
    st.text('Processing img2text...')
    scenario = img2text(uploaded_file.name)
    st.write(scenario)

    #Stage 2: Text to Story
    st.text('Generating a story...')
    #scenario = "children playing in the park illustration"
    story = text2story(scenario)
    st.write(story)

    #Stage 3: Story to Audio data
    st.text('Generating audio data...')
    audio_data =text2audio(story)

    # Play button
    if st.button("Play Audio"):
        st.audio(audio_data['audio'],
                    format="audio/wav",
                    start_time=0,
                    sample_rate = audio_data['sampling_rate'])