SALMONN-7B-gradio

Running on Zero

App Files Files Community

fffiloni commited on Feb 12, 2024

Commit

c5fe591

verified ·

1 Parent(s): 8608d24

Update app.py

Browse files

Files changed (1) hide show

app.py +76 -64

app.py CHANGED Viewed

@@ -59,80 +59,92 @@ def gradio_answer(speech, text_input, num_beams, temperature, top_p):
     return llm_message[0]
-title = """<h1 align="center">SALMONN: Speech Audio Language Music Open Neural Network</h1>"""
 image_src = """<h1 align="center"><a href="https://github.com/bytedance/SALMONN"><img src="https://raw.githubusercontent.com/bytedance/SALMONN/main/resource/salmon.png", alt="SALMONN" border="0" style="margin: 0 auto; height: 200px;" /></a> </h1>"""
-description = """<h3>This is the demo of SALMONN-7B. To experience SALMONN-13B, you can go to <a href="https://bytedance.github.io/SALMONN">https://bytedance.github.io/SALMONN</a>.\n Upload your audio and start chatting!</h3>"""
-with gr.Blocks() as demo:
-    gr.Markdown(title)
-    gr.Markdown(image_src)
-    gr.Markdown(description)
-    with gr.Row():
-        with gr.Column():
-            speech = gr.Audio(label="Audio", type='filepath')
-            num_beams = gr.Slider(
-                minimum=1,
-                maximum=10,
-                value=4,
-                step=1,
-                interactive=True,
-                label="beam search numbers",
-            )
-            top_p = gr.Slider(
-                minimum=0.1,
-                maximum=1.0,
-                value=0.9,
-                step=0.1,
-                interactive=True,
-                label="top p",
-            )
-            temperature = gr.Slider(
-                minimum=0.8,
-                maximum=2.0,
-                value=1.0,
-                step=0.1,
-                interactive=False,
-                label="temperature",
             )
-        with gr.Column():
-            text_input = gr.Textbox(label='User', placeholder='Please upload your audio first', interactive=True)
-            answer = gr.Textbox(label="Salmonn answer")
-    with gr.Row():
-        examples = gr.Examples(
-            examples = [
-                ["resource/audio_demo/gunshots.wav", "Recognize the speech and give me the transcription."],
-                ["resource/audio_demo/gunshots.wav", "Listen to the speech and translate it into German."],
-                ["resource/audio_demo/gunshots.wav", "Provide the phonetic transcription for the speech."],
-                ["resource/audio_demo/gunshots.wav", "Please describe the audio."],
-                ["resource/audio_demo/gunshots.wav", "Recognize what the speaker says and describe the background audio at the same time."],
-                ["resource/audio_demo/gunshots.wav", "Use your strong reasoning skills to answer the speaker's question in detail based on the background sound."],
-                ["resource/audio_demo/duck.wav", "Please list each event in the audio in order."],
-                ["resource/audio_demo/duck.wav", "Based on the audio, write a story in detail. Your story should be highly related to the audio."],
-                ["resource/audio_demo/duck.wav", "How many speakers did you hear in this audio? Who are they?"],
-                ["resource/audio_demo/excitement.wav", "Describe the emotion of the speaker."],
-                ["resource/audio_demo/mountain.wav", "Please answer the question in detail."],
-                ["resource/audio_demo/jobs.wav", "Give me only three keywords of the text. Explain your reason."],
-                ["resource/audio_demo/2_30.wav", "What is the time mentioned in the speech?"],
-                ["resource/audio_demo/music.wav", "Please describe the music in detail."],
-                ["resource/audio_demo/music.wav", "What is the emotion of the music? Explain the reason in detail."],
-                ["resource/audio_demo/music.wav", "Can you write some lyrics of the song?"],
-                ["resource/audio_demo/music.wav", "Give me a title of the music based on its rhythm and emotion."]
-            ],
-            inputs=[speech, text_input]
-        )
     text_input.submit(
         gradio_answer, [speech, text_input, num_beams, temperature, top_p], [answer]
     )
 # demo.launch(share=True, enable_queue=True, server_port=int(args.port))

     return llm_message[0]
+title = """<h1 style="text-align: center;">SALMONN: Speech Audio Language Music Open Neural Network</h1>"""
 image_src = """<h1 align="center"><a href="https://github.com/bytedance/SALMONN"><img src="https://raw.githubusercontent.com/bytedance/SALMONN/main/resource/salmon.png", alt="SALMONN" border="0" style="margin: 0 auto; height: 200px;" /></a> </h1>"""
+description = """<h3 style="text-align: center;">This is the simplified demo for SALMONN-7B. To experience SALMONN-13B, you can go to <a href="https://bytedance.github.io/SALMONN">https://bytedance.github.io/SALMONN</a>.\n Upload your audio and ask a question!</h3>"""
+css = """
+div#col-container {
+    margin: 0 auto;
+    max-width: 840px;
+}
+"""
+with gr.Blocks(css=css) as demo:
+    with gr.Column(elem_id="col-container"):
+        gr.Markdown(title)
+        #gr.Markdown(image_src)
+        gr.Markdown(description)
+        with gr.Row():
+            with gr.Column():
+                speech = gr.Audio(label="Audio", type='filepath')
+                with gr.Accordion("Advanced Settings", open=False):
+                    num_beams = gr.Slider(
+                        minimum=1,
+                        maximum=10,
+                        value=4,
+                        step=1,
+                        interactive=True,
+                        label="beam search numbers",
+                    )
+                    top_p = gr.Slider(
+                        minimum=0.1,
+                        maximum=1.0,
+                        value=0.9,
+                        step=0.1,
+                        interactive=True,
+                        label="top p",
+                    )
+                    temperature = gr.Slider(
+                        minimum=0.8,
+                        maximum=2.0,
+                        value=1.0,
+                        step=0.1,
+                        interactive=False,
+                        label="temperature",
+                    )
+            with gr.Column():
+                with gr.Row():
+                    text_input = gr.Textbox(label='User question', placeholder='Please upload your audio first', interactive=True)
+                    submit_btn = gr.Button("Submit")
+                answer = gr.Textbox(label="Salmonn answer")
+        with gr.Row():
+            examples = gr.Examples(
+                examples = [
+                    ["resource/audio_demo/gunshots.wav", "Recognize the speech and give me the transcription."],
+                    ["resource/audio_demo/gunshots.wav", "Listen to the speech and translate it into German."],
+                    ["resource/audio_demo/gunshots.wav", "Provide the phonetic transcription for the speech."],
+                    ["resource/audio_demo/gunshots.wav", "Please describe the audio."],
+                    ["resource/audio_demo/gunshots.wav", "Recognize what the speaker says and describe the background audio at the same time."],
+                    ["resource/audio_demo/gunshots.wav", "Use your strong reasoning skills to answer the speaker's question in detail based on the background sound."],
+                    ["resource/audio_demo/duck.wav", "Please list each event in the audio in order."],
+                    ["resource/audio_demo/duck.wav", "Based on the audio, write a story in detail. Your story should be highly related to the audio."],
+                    ["resource/audio_demo/duck.wav", "How many speakers did you hear in this audio? Who are they?"],
+                    ["resource/audio_demo/excitement.wav", "Describe the emotion of the speaker."],
+                    ["resource/audio_demo/mountain.wav", "Please answer the question in detail."],
+                    ["resource/audio_demo/jobs.wav", "Give me only three keywords of the text. Explain your reason."],
+                    ["resource/audio_demo/2_30.wav", "What is the time mentioned in the speech?"],
+                    ["resource/audio_demo/music.wav", "Please describe the music in detail."],
+                    ["resource/audio_demo/music.wav", "What is the emotion of the music? Explain the reason in detail."],
+                    ["resource/audio_demo/music.wav", "Can you write some lyrics of the song?"],
+                    ["resource/audio_demo/music.wav", "Give me a title of the music based on its rhythm and emotion."]
+                ],
+                inputs=[speech, text_input]
             )
     text_input.submit(
         gradio_answer, [speech, text_input, num_beams, temperature, top_p], [answer]
     )
+    submit_btn.click(
+        gradio_answer, [speech, text_input, num_beams, temperature, top_p], [answer]
+    )
 # demo.launch(share=True, enable_queue=True, server_port=int(args.port))