File size: 12,505 Bytes
37534d8
67c5640
96385e2
239a0a9
 
96385e2
67c5640
 
5282ab0
 
a9c9ab3
 
5282ab0
a9c9ab3
5282ab0
 
 
 
 
 
 
a9c9ab3
 
 
 
 
 
 
 
5282ab0
 
 
 
 
 
a9c9ab3
 
 
 
 
 
 
 
 
 
 
 
67c5640
 
239a0a9
c9de9e8
239a0a9
 
37534d8
 
 
 
96385e2
37534d8
96385e2
 
 
 
 
 
 
37534d8
726cc64
 
 
 
96385e2
726cc64
c9de9e8
96385e2
 
 
726cc64
b988012
726cc64
96385e2
726cc64
37534d8
67c5640
726cc64
b988012
37534d8
 
 
 
 
 
 
 
 
 
 
 
96385e2
37534d8
96385e2
 
 
 
 
 
37534d8
a9c9ab3
37534d8
96385e2
6e834ca
88080c5
 
 
 
 
 
239a0a9
 
 
 
b988012
239a0a9
 
 
 
 
 
 
88080c5
239a0a9
88080c5
a9c9ab3
 
239a0a9
b988012
239a0a9
 
 
 
6e834ca
239a0a9
 
 
 
 
 
c9de9e8
239a0a9
 
c9de9e8
239a0a9
 
 
 
 
 
 
 
 
 
 
 
c9de9e8
239a0a9
 
b988012
 
 
239a0a9
 
 
b988012
239a0a9
 
 
 
 
b988012
239a0a9
 
 
 
b988012
239a0a9
 
b988012
239a0a9
b988012
239a0a9
 
 
b988012
239a0a9
 
 
c9de9e8
239a0a9
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
c9de9e8
6e834ca
 
239a0a9
 
b988012
239a0a9
b988012
239a0a9
b988012
239a0a9
 
 
 
 
 
88080c5
239a0a9
 
 
 
 
a9c9ab3
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
import json
import os
import re
import time
import pandas as pd
import gradio as gr
from google import genai
from google.genai import types
from huggingface_hub import hf_hub_download, HfApi

# 雲端 Dataset 設定
REPO_ID = "Denny1911/Keywords"  
FILENAME = "tags.json"
HF_TOKEN = os.environ.get("HF_TOKEN") 

def load_tags_from_cloud():
    """從雲端 Dataset 下載最新 JSON"""
    try:
        filepath = hf_hub_download(repo_id=REPO_ID, filename=FILENAME, repo_type="dataset", token=HF_TOKEN)
        with open(filepath, "r", encoding="utf-8") as f:
            return json.load(f)
    except Exception as e:
        print(f"讀取雲端失敗,使用預設初始標籤。原因: {e}")
        return {
            "情緒基調": ["溫柔療癒", "熱血激情"],
            "視覺畫面": ["璀璨星空"],
            "適用場景": ["深夜沉思"],
            "風格曲風": ["華語流行"]
        }

def save_tags_to_cloud(tags_data):
    """將更新後的 JSON 直接推回雲端 Dataset 儲存"""
    with open(FILENAME, "w", encoding="utf-8") as f:
        json.dump(tags_data, f, ensure_ascii=False, indent=2)
    
    try:
        api = HfApi()
        api.upload_file(
            path_or_fileobj=FILENAME,
            path_in_repo=FILENAME,
            repo_id=REPO_ID,
            repo_type="dataset",
            token=HF_TOKEN
        )
        print("雲端標籤庫同步成功!")
    except Exception as e:
        print(f"同步至雲端失敗。原因: {e}")


# =====================================================================
# 核心 AI 分析邏輯
# =====================================================================
def core_analyze_process(artist, song, current_tags):
    client = genai.Client(api_key=os.environ.get("GEMINI_API_KEY"))
    
    prompt = f"""
    你是一位專業的音樂與情感分析師。
    請針對用戶提供的【歌曲:{song}】與【歌手:{artist}】,利用聯網搜尋功能找到這首歌的完整背景、歌詞與音樂資訊。
    
    請嚴格按照以下格式輸出結果(包含兩個區塊):

    ### 📝 1. 意境與情境描述
    [請用 150 字以內,精準、優美地簡述這首歌在描述什麼樣的情境、故事或情感。]

    ### 🏷️ 2. 統一標籤庫
    [請將這首歌的所有相關標籤,全部統一用「英文逗號 ,」隔開,排成一整行輸出,不要分段。]
    
    ✨【中英雙語對照輸出規則】✨:
    1. 只要輸出了「中文/非英文」的標籤,其後方必須「緊接著」輸出它的英文翻譯,並用逗號隔開。
    2. 如果標籤本身就是數字(如發行年份 2007、速度 135 BPM)或是原本就屬於英文單字(如 Taiwan, 2000s),則保持原樣即可,絕對不要重複翻譯。
    
    字串中必須包含以下兩大類標籤:
    1. 靜態標籤(需遵循雙語對照規則):
       歌曲名、專輯或歌曲衍生名、發行年份、年代標籤(如 2000年代, 2000s)、地區、相關影視作品名稱、最著名的副歌第一句歌詞、歌曲速度(如 135 BPM)。
       
    2. 動態標籤(必須對照下方的【標準資料庫】進行語意對齊):
       包含【情緒基調】、【視覺畫面】、【適用場景】、【風格曲風】。
       - 語意對齊:如果你想到的詞與下方資料庫中的詞高度相似,必須自動校正並採用資料庫中的詞(採用後一樣要在後面補上英文翻譯)。
       - 發現新詞規則:如果你認為現有資料庫無法涵蓋,允许你創造新詞,但必須在「中文新詞」前面加上 `[NEW:類別]` 標記,並在其後方照常輸出英文翻譯。
         (例如:創造新曲風寫成 `[NEW:風格曲風]周氏情歌, Chou Style`;創造新情感寫成 `[NEW:情緒基調]青澀回憶, Youthful Memories`)。

    【目前的標準資料庫】(供你對齊與參考語意):
    {json.dumps(current_tags, ensure_ascii=False)}

    正確的【統一標籤庫】中英交錯輸出範例(必須只有一行長字串,不要帶引號):
    蒲公英的約定, The Promised Dandelion, 我box忙, On the Run, 2007, 2000年代, 2000s, 台灣, Taiwan, 不能說的秘密, Secret, 一起長大的約定 那樣清晰 打過勾的我相信, The promise of growing up together is so clear, 135 BPM, [NEW:風格曲風]周氏情歌, Chou Style, 溫柔療癒, Gentle and Healing, 璀璨星空, Starry Night, 深夜沉思, Midnight Reflection
    """

    response = client.models.generate_content(
        model='gemini-2.5-flash',
        contents=prompt,
        config=types.GenerateContentConfig(
            tools=[types.Tool(google_search=types.GoogleSearch())]
        )
    )
    
    result_text = response.text
    
    new_found_tags = re.findall(r'\[NEW:(情緒基調|視覺畫面|適用場景|風格曲風)\]([^,\n]+)', result_text)
    has_updates = False
    for category, tag in new_found_tags:
        tag = tag.strip()
        if tag not in current_tags[category]:
            current_tags[category].append(tag)
            has_updates = True
            
    if has_updates:
        save_tags_to_cloud(current_tags)
        
    clean_output = re.sub(r'\[NEW:[^\]]+\]', '', result_text)
    
    desc_match = re.search(r'### 📝 1\. 意境與情境描述\s*([\s\S]*?)(?=### 🏷️ 2\. 統一標籤庫|$)', clean_output)
    tags_match = re.search(r'### 🏷️ 2\. 統一標籤庫\s*([\s\S]*)', clean_output)
    
    desc_content = desc_match.group(1).strip() if desc_match else "(未能成功生成情境描述)"
    tags_content = tags_match.group(1).strip() if tags_match else clean_output
    
    return tags_content, desc_content


# =====================================================================
# 功能 1:單曲分析(保留前台意境分流呈現)
# =====================================================================
def analyze_song_single(artist, song, progress=gr.Progress()):
    progress(0.1, desc="🔄 正在連線雲端下載歷史標籤庫...")
    current_tags = load_tags_from_cloud()
    
    progress(0.3, desc=f"🔍 正在啟動 Gemini 聯網檢索《{song}》...")
    tags_content, desc_content = core_analyze_process(artist, song, current_tags)
    
    progress(1.0, desc="🎉 單曲分析大功告成!")
    return tags_content, desc_content


# =====================================================================
# 功能 2:批次表格上傳分析(✨ 嚴格限制三直欄輸出,移除非必要欄位)
# =====================================================================
def analyze_song_batch(file_obj, delay_seconds, progress=gr.Progress()):
    if file_obj is None:
        return None, "❌ 請先上傳 Excel 或 CSV 檔案!"
    
    try:
        if file_obj.name.endswith('.csv'):
            df = pd.read_csv(file_obj.name)
        else:
            df = pd.read_excel(file_obj.name)
    except Exception as e:
        return None, f"❌ 檔案讀取失敗。原因: {e}"
    
    if df.shape[1] < 2:
        return None, "❌ 檔案格式不符!表格必須至少包含兩欄(左邊歌曲,右邊歌手)。"
    
    total_songs = len(df)
    results = []
    
    progress(0.05, desc="🔄 正在連線雲端下載歷史標籤庫...")
    current_tags = load_tags_from_cloud()
    
    for index, row in df.iterrows():
        song = str(row.iloc[0]).strip()
        artist = str(row.iloc[1]).strip()
        
        current_idx = index + 1
        progress((index / total_songs), desc=f"🎵 [批次進度 {current_idx}/{total_songs}] 正在分析: {artist} - 《{song}》...")
        
        try:
            # 呼叫 AI 獲取標籤(忽略回傳的 desc 內容)
            tags, _ = core_analyze_process(artist, song, current_tags)
            # ✨ 核心優化:字典裡只塞三個直欄欄位,徹底移除意境描述
            results.append({
                "歌曲": song,
                "歌手": artist,
                "雙語標籤庫": tags
            })
        except Exception as e:
            results.append({
                "歌曲": song,
                "歌手": artist,
                "雙語標籤庫": f"分析失敗: {e}"
            })
        
        if current_idx < total_songs and delay_seconds > 0:
            for remaining in range(int(delay_seconds), 0, -1):
                progress((current_idx / total_songs), desc=f"⏳ [防封鎖冷卻中] 已完成第 {current_idx} 首。等待 {remaining} 秒後繼續下一首...")
                time.sleep(1)

    progress(0.95, desc="💾 所有歌曲分析完畢!正在封裝成精簡版 Excel 檔案...")
    output_df = pd.DataFrame(results)
    output_filename = "音樂批次分析結果_精簡雙語版.xlsx"
    output_df.to_excel(output_filename, index=False)
    
    progress(1.0, desc="🎉 批次分析全部完成!檔案已生成。")
    return output_filename, f"✅ 成功處理完成!共計 {total_songs} 首歌曲,請點擊下方按鈕下載僅含三直欄的完美表格。"


# =====================================================================
# Gradio 介面佈局
# =====================================================================
with gr.Blocks(title="🎵 智慧音樂關鍵字大數據分析系統") as demo:
    gr.Markdown("# 🎵 智慧音樂關鍵字與意境分析系統 (全功能版)")
    
    with gr.Tabs():
        # --- 分頁一:單曲即時分析 ---
        with gr.Tab("🎯 單曲即時分析"):
            gr.Markdown("輸入單一歌手與歌名,AI 自動聯網檢索,即時呈現雙語標籤海。")
            with gr.Row():
                with gr.Column():
                    artist_input = gr.Textbox(label="歌手 / 藝術家", placeholder="例如:周杰倫")
                    song_input = gr.Textbox(label="歌曲名稱", placeholder="例如:蒲公英的約定")
                    with gr.Row():
                        clear_btn = gr.Button("Clear")
                        submit_btn = gr.Button("Submit", variant="primary")
                with gr.Column():
                    gr.Markdown("### 🏷️ 智慧生成雙語標籤庫")
                    output_tags = gr.Markdown()
                    with gr.Accordion("📝 點擊展開:查看 AI 歌曲意境與情境描述", open=False):
                        output_desc = gr.Markdown()
            
            submit_btn.click(
                fn=analyze_song_single,
                inputs=[artist_input, song_input],
                outputs=[output_tags, output_desc]
            )
            clear_btn.click(
                fn=lambda: ("", "", "", ""),
                inputs=None,
                outputs=[artist_input, song_input, output_tags, output_desc]
            )

        # --- 分頁二:批次表格上傳 ---
        with gr.Tab("📂 批次表格上傳分析"):
            gr.Markdown("### 📊 多筆資料自動化分析")
            gr.Markdown("請上傳一個 Excel (.xlsx) 或 CSV (.csv) 檔案。第一欄為「歌曲名稱」,第二欄為「歌手/團隊」。")
            
            with gr.Row():
                with gr.Column():
                    file_input = gr.File(label="上傳音樂表格檔案", file_types=[".xlsx", ".csv"])
                    # ✨ 核心優化:將預設間隔時間拉高到 12 秒,並擴大極限至 30 秒,防止大量查詢被 Google 判定為惡意刷流量
                    delay_slider = gr.Slider(
                        minimum=0, maximum=30, value=12, step=1, 
                        label="⏳ 每筆分析之間的時間間隔 (秒)", 
                        info="💡 建議維持 12 秒以上。由於免費版 Gemini 加上聯網搜尋有嚴格的每分鐘次數限制 (Rate Limit),間隔太短會導致後面整排分析失敗。"
                    )
                    batch_submit_btn = gr.Button("🚀 開始批次自動分析", variant="primary")
                
                with gr.Column():
                    batch_status = gr.Textbox(label="執行狀態報告", placeholder="等待上傳檔案...", interactive=False)
                    file_output = gr.File(label="📥 下載完整分析 Excel 結果報告")
            
            batch_submit_btn.click(
                fn=analyze_song_batch,
                inputs=[file_input, delay_slider],
                outputs=[file_output, batch_status]
            )

if __name__ == "__main__":
    demo.launch()