Ray1ee01 commited on
Commit
273b8b1
·
verified ·
1 Parent(s): 1569336

Upload folder using huggingface_hub

Browse files
Files changed (31) hide show
  1. icon_generation/backup/accepted_domains.txt +346 -0
  2. icon_generation/backup/accepted_domains_attributes.txt +0 -0
  3. icon_generation/backup/analyze_domains.py +169 -0
  4. icon_generation/backup/check_progress.sh +38 -0
  5. icon_generation/backup/domain_summary.txt +514 -0
  6. icon_generation/backup/domain_value_pairs_enhanced.txt +0 -0
  7. icon_generation/backup/domain_value_pairs_filtered.txt +0 -0
  8. icon_generation/backup/enhance_domains.py +551 -0
  9. icon_generation/backup/enhance_log.txt +1130 -0
  10. icon_generation/backup/extract_accepted_domains.py +171 -0
  11. icon_generation/backup/filter_accepted_domains.py +445 -0
  12. icon_generation/backup/filter_accepted_log.txt +383 -0
  13. icon_generation/backup/filter_pairs.py +402 -0
  14. icon_generation/backup/filtered.json +0 -0
  15. icon_generation/backup/image_batch_generator.py +571 -0
  16. icon_generation/backup/image_batch_generator_backup.py +511 -0
  17. icon_generation/backup/image_generator.py +64 -0
  18. icon_generation/backup/json_to_txt.py +77 -0
  19. icon_generation/backup/refine_domains.py +537 -0
  20. icon_generation/backup/refined_domains.json +0 -0
  21. icon_generation/backup/summarize_domains.py +142 -0
  22. icon_generation/backup/topic_style.json +520 -0
  23. icon_generation/batch_icon_generator.py +267 -0
  24. icon_generation/count.py +52 -0
  25. icon_generation/domain_attributes.txt +957 -0
  26. icon_generation/image_gen.py +119 -0
  27. icon_generation/refine_domains.py +537 -0
  28. icon_generation/split_icon.py +487 -0
  29. icon_generation/template.json +14 -0
  30. icon_generation/template_batch.json +21 -0
  31. icon_generation/test_split.py +56 -0
icon_generation/backup/accepted_domains.txt ADDED
@@ -0,0 +1,346 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Industry Sector
2
+ Web Browser
3
+ Entertainment Genre
4
+ Media Format
5
+ Smart Home Device Category
6
+ Art Medium
7
+ Product Category
8
+ Land Use Type
9
+ Device Type
10
+ Crop Type
11
+ Transportation Mode
12
+ Smartphone Brand
13
+ Vehicle Type
14
+ Energy Technologies
15
+ Energy Source
16
+ Geographic Region
17
+ Academic Subject
18
+ Material
19
+ Product Brand
20
+ Client Engagement Channel
21
+ Transportation Infrastructure Type
22
+ Physical Activity
23
+ Pet Category
24
+ Clothing Type
25
+ Dance Discipline
26
+ Musical Instruments
27
+ Building Type
28
+ Household Appliance
29
+ Incident Type
30
+ Occupation
31
+ Cuisine
32
+ Season (Calendar & Climatic)
33
+ Media Asset
34
+ Company Name
35
+ Live Performance Type
36
+ Animal Species
37
+ Event Type
38
+ Mobile Phone Type
39
+ Accommodation Type
40
+ Operating System
41
+ Vehicle Powertrain Type
42
+ Streaming Service
43
+ Weather Type
44
+ Cooking Technique
45
+ Religious Affiliation
46
+ Sentiment
47
+ Biome Type
48
+ Playground Feature
49
+ Handicraft Category
50
+ Retail Location Type
51
+ Art and Craft Techniques
52
+ Park Type
53
+ Camera Form Factor
54
+ Residential Room Type
55
+ Business Segment (Food & Beverage)
56
+ Leisure Activity
57
+ Museum Object Type
58
+ Pollution Source or Pollutant Type
59
+ Medical Injury Type
60
+ Dietary Preference or Restriction
61
+ Dietary Protein Source
62
+ Coffee Brewing Method
63
+ Social Event Type
64
+ Purchase Channel
65
+ Investment Asset Class
66
+ Expense Category
67
+ Music Genre
68
+ Dwelling Type
69
+ Time of Day
70
+ Produce (Fruit & Vegetable)
71
+ Medical Procedure
72
+ Lighting Type
73
+ Landmark Name
74
+ Seafood Species
75
+ Travel Category
76
+ Color
77
+ Cryptocurrency Name
78
+ Food Category
79
+ Fashion Accessory Category
80
+ Tourism Type
81
+ Interest Category
82
+ Financial Incentive Type
83
+ Affectionate Gestures
84
+ Mobile OS
85
+ Medical Specialty
86
+ Health Topic
87
+ Eating Occasion
88
+ Storage Type
89
+ Coffee Beverage / Preparation Style
90
+ Environmental Impact Categories
91
+ Game Genre
92
+ Holiday Name
93
+ Museum Focus
94
+ Waste Material Type
95
+ Freight Cargo Type
96
+ Funding Source
97
+ Civil Engineering Structure Type
98
+ Recipe Ingredient
99
+ Software Application Name
100
+ Sofa Type
101
+ Sales Offer Type
102
+ Financial Product Type
103
+ Festival Theme
104
+ Camera Lens Type
105
+ Concession Stand Product
106
+ Climate Classification
107
+ Primate Species
108
+ Sports Facility Type
109
+ Finishing Position
110
+ Sustainability Category
111
+ Audio Listening Method
112
+ Travel Market Segment
113
+ Fitness Goal
114
+ Communication Channel
115
+ Tourist Attraction
116
+ Grain Type
117
+ Furniture Category
118
+ Heritage Site Type
119
+ Beverage Type
120
+ Research Publication Type
121
+ Earring Type
122
+ Common Illicit and Recreational Drugs
123
+ Occasion (Usage Scenario)
124
+ Major Animal Groups
125
+ Alcoholic Beverage Type
126
+ Establishment Type
127
+ Website Content Section
128
+ Property Flooring Type
129
+ Satellite Mission Type
130
+ Vehicle Propulsion Type
131
+ Retail Store Category
132
+ Fantasy Creature or Race
133
+ Milestone Type
134
+ Cyberattack Type
135
+ Art Subject
136
+ Citrus Variety
137
+ Food Item
138
+ Insect Common Name
139
+ Insurance Type
140
+ Venue Seating Section
141
+ Home Decor Category
142
+ Driving Environment
143
+ Wearable Accessory Type
144
+ Play Activity Type
145
+ Music Distribution Format
146
+ Sports Court Surface
147
+ Application Category (App Store)
148
+ Exercise Type
149
+ Streaming Platforms
150
+ Cultivation Environment
151
+ Sensor Type
152
+ Wine Grape Varieties and Styles
153
+ Email Service Provider
154
+ Costume Character
155
+ Livestock Species
156
+ Historical Period
157
+ Class Type
158
+ Religious Symbol
159
+ Fruit Name
160
+ Pollution Source
161
+ Dietary Food Group
162
+ Neighborhood Name
163
+ Manufacturing Step
164
+ Amusement Ride Type
165
+ Dish Type
166
+ Retailer Name
167
+ Pesticide Type (Target/Function)
168
+ Vehicle Maintenance Type
169
+ Painting Medium
170
+ Craft Supplies
171
+ Project Phase
172
+ Dominant Forest Cover Type
173
+ Device Capabilities (Sensors, Connectivity & Features)
174
+ Competition Gender Division
175
+ Irrigation Source Type
176
+ Gift Category
177
+ Vessel Type
178
+ Notable Sacred Sites
179
+ Online Video Platforms
180
+ Tropical Cyclone Intensity Category (Saffir–Simpson and related classifications)
181
+ Internet Connectivity Status
182
+ Water Supply Type
183
+ Performance Terrain Type
184
+ Workout Modality
185
+ Common Pest Type
186
+ Patrol Method
187
+ Sustainable Building Feature
188
+ Fatal Incident Type
189
+ Learning Resource Type
190
+ Service Category
191
+ Manufacturing Process
192
+ Pest Control Technique
193
+ Personal Financial Concern Category
194
+ Software Application
195
+ Result Status (Test/Operation)
196
+ Weightlifting and Strength Training Exercises
197
+ Access Level
198
+ Therapeutic Area
199
+ Employment Type
200
+ Event Category
201
+ Wellness Program Type
202
+ Packaging Material
203
+ ADAS Feature
204
+ Military Platform Type
205
+ Subscription Plan Tier
206
+ Organization Sector
207
+ Planetary Rover Name
208
+ Project Category
209
+ Footwear Style
210
+ Commercial Aircraft Model
211
+ Forage Type
212
+ Clinical Service Type
213
+ Hat Style
214
+ U.S. Military Service Branch
215
+ App Permission
216
+ News Organization
217
+ Ranching Practices and Systems
218
+ Construction & Infrastructure Project Type
219
+ Home Feature
220
+ Production Method
221
+ Pricing Basis
222
+ Booking Channel
223
+ Educational Focus Area
224
+ Ecosystem / Habitat Type
225
+ Craft Materials
226
+ Subsystem / Component Type
227
+ Flavor Profile (Tasting Descriptors)
228
+ Transport Route Type
229
+ Basic Human Needs
230
+ Fashion Style
231
+ Humanitarian Aid Sector
232
+ Character Type
233
+ Aircraft Category (type/market segment)
234
+ Technology Solution Type
235
+ Roofing Material/Type
236
+ Common Amphibian Species (common names)
237
+ Mammal Species Name
238
+ Insect Group (common names)
239
+ Movie Theater Format
240
+ Food Item Name
241
+ Medical Treatment Type
242
+ Handbag Style
243
+ Artwork Title
244
+ Personal Protective Equipment (PPE) Type
245
+ Academic Subject Category
246
+ Microphone Type
247
+ Pollinator Types
248
+ Computer Peripherals
249
+ Agricultural Activities
250
+ Mission Phase
251
+ Certification Level
252
+ Leisure Amenities
253
+ Personal Relationship Type
254
+ Astronomical Observatory Name
255
+ Cultural Institution Type
256
+ Historic Site Name
257
+ Medical Interventions
258
+ Loyalty Program Feature
259
+ Racing Circuit Type
260
+ Competitor Brand (Fashion & Apparel)
261
+ Hazard Type
262
+ Pet Service Type
263
+ Pet Wellness Package Type
264
+ Benefit Recipient Type
265
+ Forms of Folklore
266
+ Generational Cohort
267
+ Vehicle Part Type
268
+ Ad Format
269
+ Jewelry-making Technique
270
+ Cultural Attire
271
+ Chinese New Year Foods
272
+ Maize (Corn) Cultivar/Variety
273
+ Consumer Electronics & Computer Hardware Category
274
+ Educational Program Type
275
+ Art Installation Type
276
+ Tour Activity
277
+ Server Role
278
+ Emotion Type
279
+ Cause of Damage
280
+ Passenger Persona (Travel Behavior)
281
+ Solar System Planet
282
+ Healthcare Facility Type
283
+ Application Functionality
284
+ Role-Playing Game System
285
+ Freight Transport Mode
286
+ Product Variant (Production Method)
287
+ Personal Data Type
288
+ Financial Transaction Type
289
+ Spending Occasion
290
+ Home Improvement Project Type
291
+ Cultural Art Traditions
292
+ Ticket Category
293
+ Hardware Component Type
294
+ Manufacturing Machine Type
295
+ Motorcycle Type
296
+ Light Color (illumination)
297
+ Endorsement Category (Products & Services)
298
+ Communication Purpose
299
+ Motor Vehicle Collision Type
300
+ Cultural Offering Type
301
+ Donor Classification
302
+ Lemur Taxon Name (Genus and Species)
303
+ Therapy Modality (Physical, Psychological, and Complementary Therapies)
304
+ Body Shape (Figure Type)
305
+ Comic Book Series
306
+ Outdoor Lighting Fixture Type
307
+ Material Sourcing Type
308
+ Power Plant
309
+ Driving Scenario
310
+ Skilled Trade
311
+ Land Use / Development Type
312
+ Ritual Type
313
+ High-Risk Patient Groups
314
+ Subject Area (Topic)
315
+ Exhibit Theme
316
+ Venue Type
317
+ Toy Type
318
+ Harvesting Method
319
+ Seabird Species
320
+ Hazard Mitigation Project Type
321
+ Accessibility Feature Type
322
+ Tractor Powertrain & Control Configuration
323
+ Filtration Technology
324
+ Running Shoe Segment
325
+ Art & Craft Workshop Discipline
326
+ Scientific Discovery Category
327
+ Monetization Method
328
+ Debris Material
329
+ Fishing Capture Method
330
+ Land Management Practice
331
+ Basketball Shot Zone
332
+ Dress Silhouette
333
+ Water Conservation Measure
334
+ Love Language Type
335
+ Crop Pest
336
+ Gift Type
337
+ Farm Diversification Strategies
338
+ Payment Method
339
+ Climate Zone (major types — Köppen & common names)
340
+ Architectural Style
341
+ Schedule Disruption Reason
342
+ Tomato Variety
343
+ Lighting Fixture Type
344
+ Washing Machine Type
345
+ Art Supply Set Type
346
+ Gender Identity
icon_generation/backup/accepted_domains_attributes.txt ADDED
The diff for this file is too large to render. See raw diff
 
icon_generation/backup/analyze_domains.py ADDED
@@ -0,0 +1,169 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """
3
+ 分析 domain_value_pairs_filtered.txt
4
+ 提取所有唯一的 domain 并统计信息
5
+ """
6
+
7
+ from collections import defaultdict
8
+ from typing import List, Dict, Tuple
9
+
10
+
11
+ def parse_csv_line(line: str) -> Tuple[str, str, int]:
12
+ """
13
+ 解析 CSV 行,处理可能包含逗号的带引号字段
14
+
15
+ Returns:
16
+ (domain, specific_attribute, count)
17
+ """
18
+ parts = []
19
+ current = []
20
+ in_quotes = False
21
+
22
+ for char in line:
23
+ if char == '"':
24
+ in_quotes = not in_quotes
25
+ elif char == ',' and not in_quotes:
26
+ parts.append(''.join(current).strip())
27
+ current = []
28
+ else:
29
+ current.append(char)
30
+ parts.append(''.join(current).strip())
31
+
32
+ if len(parts) >= 3:
33
+ domain = parts[0]
34
+ attribute = parts[1]
35
+ count = int(parts[2])
36
+ return domain, attribute, count
37
+ else:
38
+ return None, None, 0
39
+
40
+
41
+ def analyze_domains(file_path: str):
42
+ """分析文件并提取 domain 信息"""
43
+
44
+ print("=" * 80)
45
+ print("Domain 分析工具")
46
+ print("=" * 80)
47
+ print()
48
+
49
+ print(f"📖 读取文件: {file_path}")
50
+
51
+ # 统计信息
52
+ domain_stats = defaultdict(lambda: {
53
+ 'count': 0,
54
+ 'total_value_count': 0,
55
+ 'attributes': [],
56
+ 'top_attributes': []
57
+ })
58
+
59
+ total_lines = 0
60
+
61
+ # 读取文件
62
+ with open(file_path, 'r', encoding='utf-8') as f:
63
+ for line in f:
64
+ line = line.strip()
65
+ if not line:
66
+ continue
67
+
68
+ domain, attribute, count = parse_csv_line(line)
69
+ if domain:
70
+ total_lines += 1
71
+ domain_stats[domain]['count'] += 1
72
+ domain_stats[domain]['total_value_count'] += count
73
+ domain_stats[domain]['attributes'].append((attribute, count))
74
+
75
+ print(f"✅ 成功读取 {total_lines} 行数据")
76
+ print()
77
+
78
+ # 排序 domains(按 total_value_count 降序)
79
+ sorted_domains = sorted(
80
+ domain_stats.items(),
81
+ key=lambda x: x[1]['total_value_count'],
82
+ reverse=True
83
+ )
84
+
85
+ # 输出统计信息
86
+ print("=" * 80)
87
+ print("📊 统计信息:")
88
+ print("=" * 80)
89
+ print(f"唯一 Domain 数量: {len(sorted_domains)}")
90
+ print(f"总 Pairs 数量: {total_lines}")
91
+ print(f"总 Count 数: {sum(stats['total_value_count'] for _, stats in sorted_domains):,}")
92
+ print()
93
+
94
+ # 输出所有 domains
95
+ print("=" * 80)
96
+ print("📋 所有 Domains 列表 (按 total_count 降序):")
97
+ print("=" * 80)
98
+ print()
99
+
100
+ for idx, (domain, stats) in enumerate(sorted_domains, 1):
101
+ num_attributes = stats['count']
102
+ total_count = stats['total_value_count']
103
+
104
+ # 获取 top 3 attributes
105
+ top_attrs = sorted(stats['attributes'], key=lambda x: x[1], reverse=True)[:3]
106
+ top_attrs_str = ', '.join([f"{attr} ({cnt:,})" for attr, cnt in top_attrs])
107
+
108
+ print(f"{idx:3d}. {domain}")
109
+ print(f" - Attributes 数量: {num_attributes}")
110
+ print(f" - 总 Count: {total_count:,}")
111
+ print(f" - Top Attributes: {top_attrs_str}")
112
+ print()
113
+
114
+ print("=" * 80)
115
+
116
+ # 输出简洁的 domain 列表
117
+ print("\n📝 简洁 Domain 列表:")
118
+ print("=" * 80)
119
+ domain_list = [domain for domain, _ in sorted_domains]
120
+ for i in range(0, len(domain_list), 3):
121
+ line_domains = domain_list[i:i+3]
122
+ print(" " + " | ".join(f"{d:30s}" for d in line_domains))
123
+
124
+ print()
125
+ print("=" * 80)
126
+
127
+ # 保存到文件
128
+ output_file = 'domains_list.txt'
129
+ with open(output_file, 'w', encoding='utf-8') as f:
130
+ f.write("所有 Domains 列表 (按 total_count 降序)\n")
131
+ f.write("=" * 80 + "\n\n")
132
+
133
+ for idx, (domain, stats) in enumerate(sorted_domains, 1):
134
+ f.write(f"{idx}. {domain}\n")
135
+ f.write(f" Attributes: {stats['count']}, Total Count: {stats['total_value_count']:,}\n")
136
+
137
+ # Top 5 attributes
138
+ top_attrs = sorted(stats['attributes'], key=lambda x: x[1], reverse=True)[:5]
139
+ f.write(f" Top Attributes:\n")
140
+ for attr, cnt in top_attrs:
141
+ f.write(f" - {attr}: {cnt:,}\n")
142
+ f.write("\n")
143
+
144
+ print(f"💾 详细信息已保存到: {output_file}")
145
+
146
+ # 输出 Python 列表格式
147
+ print("\n🐍 Python 列表格式:")
148
+ print("=" * 80)
149
+ print("domains = [")
150
+ for domain in domain_list:
151
+ print(f" '{domain}',")
152
+ print("]")
153
+
154
+ return domain_list, domain_stats
155
+
156
+
157
+ def main():
158
+ file_path = 'domain_value_pairs_filtered.txt'
159
+ domains, stats = analyze_domains(file_path)
160
+
161
+ print()
162
+ print("=" * 80)
163
+ print("✨ 分析完成!")
164
+ print("=" * 80)
165
+
166
+
167
+ if __name__ == '__main__':
168
+ main()
169
+
icon_generation/backup/check_progress.sh ADDED
@@ -0,0 +1,38 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/bin/bash
2
+ # 检查 refine_domains.py 的处理进度
3
+
4
+ echo "========================================="
5
+ echo "Domain Refinement 进度检查"
6
+ echo "========================================="
7
+ echo ""
8
+
9
+ # 检查进程是否还在运行
10
+ if pgrep -f "refine_domains.py" > /dev/null; then
11
+ echo "✅ 脚本正在运行中..."
12
+ else
13
+ echo "⚠️ 脚本未运行"
14
+ fi
15
+
16
+ echo ""
17
+ echo "最新处理进度:"
18
+ echo "========================================="
19
+ tail -30 /home/lizhen/ChartPipeline/icon_generation/refine_log.txt | grep -E "(处理进度|已用时间|当前字段|Value filtering 完成)"
20
+
21
+ echo ""
22
+ echo "========================================="
23
+ echo "临时文件信息:"
24
+ if [ -f "/home/lizhen/ChartPipeline/icon_generation/refined_domains_temp.json" ]; then
25
+ temp_size=$(wc -l < /home/lizhen/ChartPipeline/icon_generation/refined_domains_temp.json)
26
+ echo "📄 临时文件行数: $temp_size"
27
+
28
+ # 统计已处理的字段数
29
+ processed=$(grep -c '"domain":' /home/lizhen/ChartPipeline/icon_generation/refined_domains_temp.json)
30
+ echo "✅ 已处理字段数: $processed"
31
+ else
32
+ echo "⚠️ 临时文件不存在"
33
+ fi
34
+
35
+ echo ""
36
+ echo "最后更新时间:"
37
+ ls -lh /home/lizhen/ChartPipeline/icon_generation/refine_log.txt | awk '{print $6, $7, $8}'
38
+
icon_generation/backup/domain_summary.txt ADDED
@@ -0,0 +1,514 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Industry Sector
2
+ Web Browser
3
+ Entertainment Genre
4
+ US State
5
+ Digital Platform or Channel
6
+ Media Format
7
+ Smart Home Device Category
8
+ Art Medium
9
+ Product Category
10
+ Continent
11
+ Land Use Type
12
+ Device Type
13
+ Crop Type
14
+ Transportation Mode
15
+ Major Professional Sports Leagues (Global)
16
+ Political Party
17
+ Smartphone Brand
18
+ Vehicle Type
19
+ Energy Technologies
20
+ Dog Breed
21
+ Energy Source
22
+ Irrigation Method
23
+ Geographic Region
24
+ Academic Subject
25
+ Material
26
+ Product Brand
27
+ Client Engagement Channel
28
+ Transportation Infrastructure Type
29
+ Beer Style
30
+ Physical Activity
31
+ Gemstone Type
32
+ Pet Category
33
+ Country
34
+ Clothing Type
35
+ Professional Sports Team
36
+ Dance Discipline
37
+ Musical Instruments
38
+ Airline Name
39
+ Building Type
40
+ Household Appliance
41
+ Award Medal Type
42
+ Incident Type
43
+ New York City Borough
44
+ Occupation
45
+ Cuisine
46
+ Season (Calendar & Climatic)
47
+ NCAA Division I Athletic Conference
48
+ Demographic Group
49
+ Media Asset Type
50
+ Stage Musical Title
51
+ Company Name
52
+ Live Performance Type
53
+ Art Period
54
+ Animal Species
55
+ Travel Destinations
56
+ Event Type
57
+ Mobile Phone Type
58
+ Accommodation Type
59
+ Operating System
60
+ Internet, Cloud, and Digital TV Service Providers
61
+ Vehicle Powertrain Type
62
+ Streaming Service
63
+ Superhero
64
+ Weather Type
65
+ Cooking Technique
66
+ Religious Affiliation
67
+ Sentiment
68
+ Warehouse Storage Type
69
+ Household Life Stage
70
+ Football Clubs (Association Football)
71
+ Biome Type
72
+ Family Structure
73
+ Playground Feature
74
+ Handicraft Category
75
+ Disease Name
76
+ Retail Location Type
77
+ Central Bank Name
78
+ Semiconductor Manufacturing Stage
79
+ Art and Craft Techniques
80
+ Park Type
81
+ Notable Creative Artists and Musical Acts
82
+ Roadway Classification
83
+ Nuclear Reactor Type
84
+ Bird Species
85
+ Camera Form Factor
86
+ Island Name
87
+ Residential Room Type
88
+ Business Segment (Food & Beverage)
89
+ Leisure Activity
90
+ Museum Object Type
91
+ Country / Sovereign State
92
+ Pollution Source or Pollutant Type
93
+ Medical Injury Type
94
+ Dietary Preference or Restriction
95
+ Dietary Protein Source
96
+ Park Name (Theme & Amusement Parks and Major Public Parks)
97
+ Coffee Brewing Method
98
+ Social Event Type
99
+ Purchase Channel
100
+ Family Size (number of members)
101
+ Investment Asset Class
102
+ Expense Category
103
+ Music Genre
104
+ Dwelling Type
105
+ Video Game Title
106
+ Ecosystem Type
107
+ Learning Delivery Mode
108
+ Vehicle Model
109
+ City
110
+ Time of Day
111
+ Produce (Fruit & Vegetable)
112
+ Medical Procedure
113
+ Major River Name
114
+ Lighting Type
115
+ Family Generational Role
116
+ Landmark Name
117
+ Sports Position
118
+ Seafood Species
119
+ Occupational Role
120
+ Travel Category
121
+ Marine Protected Area Name
122
+ Color
123
+ Protected Area Name
124
+ Cryptocurrency Name
125
+ Food Category
126
+ Launch Outcome
127
+ Jewelry Design Style
128
+ Fashion Accessory Category
129
+ Galaxy Type (morphology & class)
130
+ Tourism Type
131
+ Art Movement
132
+ Currency Code
133
+ U.S. National Park Name
134
+ Theatrical Show Title
135
+ Interest Category
136
+ Financial Incentive Type
137
+ Affectionate Gestures
138
+ Cryptocurrency Exchange
139
+ Mobile OS
140
+ Lake Name
141
+ Medical Specialty
142
+ Plant Establishment Method
143
+ Health Topic
144
+ Higher Education Institution Type
145
+ Eating Occasion
146
+ Storage Type
147
+ Zoning District (Land Use)
148
+ Coffee Beverage / Preparation Style
149
+ U.S. Battleground (Swing) States
150
+ Environmental Impact Categories
151
+ Education Program or Level
152
+ Data Collection Method
153
+ Game Genre
154
+ Holiday Name
155
+ Museum Focus
156
+ Waste Material Type
157
+ Agricultural Practice
158
+ Freight Cargo Type
159
+ Game Title
160
+ Funding Source
161
+ Customer Acquisition Channel
162
+ Supply Chain Stage
163
+ Civil Engineering Structure Type
164
+ Recipe Ingredient
165
+ Software Application Name
166
+ Sofa Type
167
+ Sales Offer Type
168
+ Financial Product Type
169
+ Festival Theme
170
+ Camera Lens Type
171
+ Concession Stand Product
172
+ Climate Classification
173
+ Primate Species
174
+ Sports Facility Type
175
+ Finishing Position
176
+ Sustainability Category
177
+ Innovation Focus Area
178
+ Audio Listening Method
179
+ Travel Market Segment
180
+ Fitness Goal
181
+ Communication Channel
182
+ Social Determinants of Health
183
+ Light Pollution Severity
184
+ Tourist Attraction
185
+ Grain Type
186
+ Medical Condition
187
+ Furniture Category
188
+ Heritage Site Type
189
+ Beverage Type
190
+ Photography Genre
191
+ Louisiana Parish
192
+ North American Bird Flyways
193
+ Research Publication Type
194
+ Education Level
195
+ Earring Type
196
+ Common Illicit and Recreational Drugs
197
+ Child Developmental Skill Category
198
+ Occasion (Usage Scenario)
199
+ Biomass Feedstock Type
200
+ Major Animal Groups
201
+ Alcoholic Beverage Type
202
+ Establishment Type
203
+ Interior Design Style
204
+ Website Content Section
205
+ Property Flooring Type
206
+ Health Condition Category
207
+ Membership Type
208
+ Vehicle Connected Services
209
+ Satellite Mission Type
210
+ Vehicle Propulsion Type
211
+ Marketing Channel
212
+ Outreach Channel
213
+ Retail Store Category
214
+ Building Siding Material
215
+ Fantasy Creature or Race
216
+ Data Source Type
217
+ Milestone Type
218
+ Cyberattack Type
219
+ Art Subject
220
+ Citrus Variety
221
+ Food Item
222
+ Systems of Medicine
223
+ Insect Common Name
224
+ Major Sporting Events and Championships
225
+ Insurance Type
226
+ Major Entertainment Award Ceremonies
227
+ Price Level
228
+ Environmental Mitigation Methods
229
+ Venue Seating Section
230
+ Home Decor Category
231
+ Cancer Type
232
+ Driving Environment
233
+ Wearable Accessory Type
234
+ Play Activity Type
235
+ Reef Site
236
+ Music Distribution Format
237
+ Sports Court Surface
238
+ Application Category (App Store)
239
+ Blockchain Platform
240
+ Exercise Type
241
+ Streaming Platforms
242
+ Gemstone Color
243
+ Cultivation Environment
244
+ Sensor Type
245
+ Wine Grape Varieties and Styles
246
+ Sport Competition Tier
247
+ Email Service Provider
248
+ Costume Character
249
+ Livestock Species
250
+ Vineyard/Appellation (Wine Region)
251
+ Historical Period
252
+ Class Type
253
+ Permit Sector (Construction/Development)
254
+ Religious Symbol
255
+ Level of Care (Senior / Long-Term Care)
256
+ Cost Element (Accounting / Project Cost Category)
257
+ Fruit Name
258
+ Pollution Source
259
+ Dietary Food Group
260
+ Major Agricultural Regions (global)
261
+ Neighborhood Name
262
+ Manufacturing Step
263
+ Amusement Ride Type
264
+ Gaming Publication
265
+ Dish Type
266
+ Retailer Name
267
+ Social & Environmental Impact Area
268
+ Pesticide Type (Target/Function)
269
+ Vehicle Maintenance Type
270
+ Economic Region (industrial and economic clusters)
271
+ Painting Medium
272
+ Environmental Remediation Method
273
+ Professional Skill
274
+ Marketing Campaign Type
275
+ Insulation Material
276
+ Craft Supplies
277
+ Geographic Region (Global, Continental, and Subregional)
278
+ Major Global Shipping Passages and Chokepoints
279
+ Project Phase
280
+ Comic Book Issue
281
+ Dominant Forest Cover Type
282
+ Device Capabilities (Sensors, Connectivity & Features)
283
+ Competition Gender Division
284
+ Irrigation Source Type
285
+ Gift Category
286
+ Vessel Type
287
+ Gallery Sector (Ownership / Institution Type)
288
+ Notable Sacred Sites
289
+ Online Video Platforms
290
+ Agricultural Operation Type
291
+ Tropical Cyclone Intensity Category (Saffir–Simpson and related classifications)
292
+ Internet Connectivity Status
293
+ Major Transportation Hub Cities
294
+ Water Supply Type
295
+ Performance Terrain Type
296
+ Coral Reef Region
297
+ Workout Modality
298
+ Common Pest Type
299
+ Patrol Method
300
+ Library Name (Major Public, National, and Research Libraries)
301
+ Customer Segment
302
+ Sustainable Building Feature
303
+ Subscription Action
304
+ Fatal Incident Type
305
+ Consumer Segment
306
+ Learning Resource Type
307
+ Service Category
308
+ Manufacturing Process
309
+ Pest Control Technique
310
+ Personal Financial Concern Category
311
+ Software Application
312
+ Result Status (Test/Operation)
313
+ Agricultural Support Category
314
+ Weightlifting and Strength Training Exercises
315
+ Access Level
316
+ Therapeutic Area
317
+ Employment Type
318
+ Event Category
319
+ Wellness Program Type
320
+ Packaging Material
321
+ Major Disaster Events
322
+ ADAS Feature
323
+ Military Platform Type
324
+ Subscription Plan Tier
325
+ Public Policy Areas
326
+ Organization Sector
327
+ Notable National Leaders (Heads of State or Government)
328
+ Planetary Rover Name
329
+ Prefectures of Japan
330
+ Ballet Title
331
+ Project Category
332
+ Space Mission
333
+ Footwear Style
334
+ Commercial Aircraft Model
335
+ Environmental Management Strategies
336
+ Forage Type
337
+ Clinical Service Type
338
+ Tree Pruning System
339
+ Hat Style
340
+ Zoning/Management Zone Type
341
+ U.S. Military Service Branch
342
+ App Permission
343
+ News Organization
344
+ Ranching Practices and Systems
345
+ Holding Institution or Collection
346
+ Construction & Infrastructure Project Type
347
+ Landscape Corridor or Long-Distance Trail Name
348
+ Home Feature
349
+ Production Method
350
+ Pricing Basis
351
+ Booking Channel
352
+ Educational Focus Area
353
+ Ecosystem / Habitat Type
354
+ Craft Materials
355
+ Subsystem / Component Type
356
+ Flavor Profile (Tasting Descriptors)
357
+ Transport Route Type
358
+ Conservation Organization
359
+ Basic Human Needs
360
+ Fashion Style
361
+ Humanitarian Aid Sector
362
+ Character Type
363
+ Flag State (Country of Registration)
364
+ Aircraft Category (type/market segment)
365
+ Technology Solution Type
366
+ Roofing Material/Type
367
+ Major Fishery Locations
368
+ Instructional Method
369
+ Common Amphibian Species (common names)
370
+ Mammal Species Name
371
+ Insect Group (common names)
372
+ Major U.S. Rivers
373
+ Major League Baseball Ballpark
374
+ Movie Theater Format
375
+ Major Rail and Metro Stations
376
+ Food Item Name
377
+ Medical Treatment Type
378
+ Environmental Management Technique
379
+ Handbag Style
380
+ Artwork Title
381
+ Personal Protective Equipment (PPE) Type
382
+ Academic Subject Category
383
+ Microphone Type
384
+ Space Agency
385
+ Pollinator Types
386
+ Countries Currently Affected by Armed Conflict or Political Violence
387
+ Computer Peripherals
388
+ U.S. Federal Agency
389
+ Agricultural Activities
390
+ Government Budget Category
391
+ Mission Phase
392
+ Certification Level
393
+ Leisure Amenities
394
+ Personal Relationship Type
395
+ Astronomical Observatory Name
396
+ National Capital Cities
397
+ Cultural Institution Type
398
+ Historic Site Name
399
+ Medical Interventions
400
+ Types of Environmental Regulations
401
+ Loyalty Program Feature
402
+ Racing Circuit Type
403
+ Competitor Brand (Fashion & Apparel)
404
+ Hazard Type
405
+ Voter Segment
406
+ Pet Service Type
407
+ Pet Wellness Package Type
408
+ Benefit Recipient Type
409
+ Forms of Folklore
410
+ Generational Cohort
411
+ Vehicle Part Type
412
+ Ad Format
413
+ Jewelry-making Technique
414
+ Theme Park Name
415
+ Cultural Attire
416
+ Chinese New Year Foods
417
+ Maize (Corn) Cultivar/Variety
418
+ Software Functional Area
419
+ U.S. Historic Site Name
420
+ Consumer Electronics & Computer Hardware Category
421
+ Educational Program Type
422
+ Art Installation Type
423
+ Culinary Trend
424
+ Tour Activity
425
+ International Trade Barrier Type
426
+ Server Role
427
+ Emotion Type
428
+ Cause of Damage
429
+ Passenger Persona (Travel Behavior)
430
+ Solar System Planet
431
+ Healthcare Facility Type
432
+ Application Functionality
433
+ Role-Playing Game System
434
+ Freight Transport Mode
435
+ Plantation Companies (Palm Oil & Tropical Crops)
436
+ Product Variant (Production Method)
437
+ Subsidy Program Name
438
+ Accommodation Property Name
439
+ Personal Data Type
440
+ Financial Transaction Type
441
+ Spending Occasion
442
+ Home Improvement Project Type
443
+ Consumer Market Segment
444
+ Cultural Art Traditions
445
+ Dimensions of Wellbeing
446
+ Ticket Category
447
+ Hardware Component Type
448
+ Manufacturing Machine Type
449
+ Motorcycle Type
450
+ Light Color (illumination)
451
+ Named Coasts and Coastal Regions (global)
452
+ Endorsement Category (Products & Services)
453
+ Communication Purpose
454
+ Motor Vehicle Collision Type
455
+ Cultural Offering Type
456
+ Donor Classification
457
+ Lemur Taxon Name (Genus and Species)
458
+ Therapy Modality (Physical, Psychological, and Complementary Therapies)
459
+ Body Shape (Figure Type)
460
+ Comic Book Series
461
+ Outdoor Lighting Fixture Type
462
+ Material Sourcing Type
463
+ Power Plant
464
+ Driving Scenario
465
+ Flight Route (region-to-region categories)
466
+ Skilled Trade
467
+ Magazine Name
468
+ Land Use / Development Type
469
+ Vehicle Model and Trim
470
+ Ritual Type
471
+ High-Risk Patient Groups
472
+ Consumer & Lifestyle Trends
473
+ TV Network (U.S. major broadcast & cable)
474
+ Regulatory Policy Area
475
+ Subject Area (Topic)
476
+ Exhibit Theme
477
+ Venue Type
478
+ Toy Type
479
+ Proximity Level
480
+ Harvesting Method
481
+ Smartphone Model
482
+ Seabird Species
483
+ Hazard Mitigation Project Type
484
+ Accessibility Feature Type
485
+ Tractor Powertrain & Control Configuration
486
+ Major Orchard Regions (Apple & Tree-Fruit Producing Regions)
487
+ Filtration Technology
488
+ Running Shoe Segment
489
+ Art & Craft Workshop Discipline
490
+ Scientific Discovery Category
491
+ Monetization Method
492
+ Area of Specialization
493
+ Company Size Category (by number of employees)
494
+ Debris Material
495
+ Fishing Capture Method
496
+ Land Management Practice
497
+ Basketball Shot Zone
498
+ Dress Silhouette
499
+ Water Conservation Measure
500
+ Highest Educational Attainment
501
+ Love Language Type
502
+ Crop Pest
503
+ Gift Type
504
+ Farm Diversification Strategies
505
+ Payment Method
506
+ Climate Zone (major types — Köppen & common names)
507
+ Architectural Style
508
+ Schedule Disruption Reason
509
+ Tomato Variety
510
+ Lighting Fixture Type
511
+ Washing Machine Type
512
+ Art Supply Set Type
513
+ Gender Identity
514
+ Marital Status
icon_generation/backup/domain_value_pairs_enhanced.txt ADDED
The diff for this file is too large to render. See raw diff
 
icon_generation/backup/domain_value_pairs_filtered.txt ADDED
The diff for this file is too large to render. See raw diff
 
icon_generation/backup/enhance_domains.py ADDED
@@ -0,0 +1,551 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """
3
+ 处理 domain_value_pairs_filtered.txt
4
+ 1. 按 domain 聚合
5
+ 2. 只保留 accepted_domains.txt 中的 domains
6
+ 3. 使用 LLM 补全 specific values 并按 real-world frequency 排序
7
+ 4. 丢弃过于冷门、罕见的 values
8
+ """
9
+
10
+ import json
11
+ import os
12
+ import requests
13
+ from typing import List, Dict, Tuple, Optional
14
+ from collections import defaultdict
15
+ import time
16
+ import threading
17
+ from concurrent.futures import ThreadPoolExecutor, as_completed
18
+
19
+
20
+ class ValueEnhancer:
21
+ """使用 LLM 增强和排序 values"""
22
+
23
+ def __init__(self, api_key=None, base_url=None, model=None):
24
+ """初始化 LLM analyzer"""
25
+ self.api_key = api_key or os.getenv("OPENAI_API_KEY") or os.getenv("AIHUBMIX_API_KEY", "")
26
+ self.base_url = base_url or os.getenv("OPENAI_BASE_URL", "https://aihubmix.com/v1")
27
+ self.model = model or os.getenv("OPENAI_MODEL", "gpt-5-mini")
28
+ self.lock = threading.Lock() # 线程锁,用于打印
29
+ self.completed_count = 0 # 已完成计数
30
+ self.total_domains = 0 # 总 domain 数
31
+
32
+ def enhance_and_sort_values(self, domain: str, values: List[Tuple[str, int]], idx: int, total: int, start_time: float) -> Dict:
33
+ """
34
+ 使用 LLM 增强和排序 values
35
+
36
+ Args:
37
+ domain: domain 名称
38
+ values: [(value, count), ...] 列表
39
+ idx: 当前索引
40
+ total: 总数
41
+ start_time: 开始时间
42
+
43
+ Returns:
44
+ {
45
+ 'enhanced_values': [(value, estimated_frequency), ...],
46
+ 'reasoning': 'explanation',
47
+ 'domain': domain
48
+ }
49
+ """
50
+ # 显示开始处理
51
+ with self.lock:
52
+ progress = (idx / total) * 100
53
+ elapsed = time.time() - start_time
54
+ avg_time = elapsed / idx if idx > 0 else 0
55
+ remaining = avg_time * (total - idx)
56
+
57
+ print(f"\n{'=' * 80}")
58
+ print(f"🔄 处理 {idx}/{total} ({progress:.1f}%): {domain}")
59
+ print(f"⏱️ 已用时间: {elapsed:.1f}秒 | 预计剩余: {remaining:.1f}秒")
60
+ print(f" 原始 values: {len(values)} 个")
61
+ print(f" 原始总 count: {sum(c for _, c in values):,}")
62
+
63
+ # 显示前 3 个示例
64
+ print(f" 示例值:")
65
+ for i, (value, count) in enumerate(sorted(values, key=lambda x: x[1], reverse=True)[:3]):
66
+ print(f" {i+1}. {value} ({count:,})")
67
+
68
+ print(f" 🤖 调用 LLM 进行增强和排序...")
69
+
70
+ prompt = self._build_enhancement_prompt(domain, values)
71
+
72
+ try:
73
+ response = self._query_llm(prompt)
74
+
75
+ if response:
76
+ # 清理可能的 markdown 代码块
77
+ cleaned_response = response.strip()
78
+ if cleaned_response.startswith('```'):
79
+ lines = cleaned_response.split('\n')
80
+ cleaned_response = '\n'.join(lines[1:-1] if lines[-1].strip() == '```' else lines[1:])
81
+ cleaned_response = cleaned_response.replace('```json', '').replace('```', '').strip()
82
+
83
+ result = json.loads(cleaned_response)
84
+
85
+ # 验证返回格式
86
+ if 'enhanced_values' in result:
87
+ enhanced_values = result['enhanced_values']
88
+ reasoning = result.get('reasoning', '')
89
+ refined_domain_name = result.get('refined_domain_name')
90
+
91
+ # 如果有新的 domain name,使用新的
92
+ final_domain = refined_domain_name if refined_domain_name else domain
93
+
94
+ with self.lock:
95
+ self.completed_count += 1
96
+ progress = (self.completed_count / total) * 100
97
+
98
+ # 显示 domain name 变化(如果有)
99
+ if refined_domain_name and refined_domain_name != domain:
100
+ print(f" 🔄 Domain 名称优化:")
101
+ print(f" 原始: {domain}")
102
+ print(f" 优化: {refined_domain_name}")
103
+
104
+ print(f" ✅ 增强完成!")
105
+ print(f" 增强后 values: {len(enhanced_values)} 个")
106
+ print(f" 变化: {len(enhanced_values) - len(values):+d}")
107
+ print(f" 理由: {reasoning}")
108
+
109
+ # 显示前 3 个增强后的值
110
+ print(f" 增强后 Top 3:")
111
+ for i, item in enumerate(enhanced_values[:3], 1):
112
+ value = item['value']
113
+ freq = item['estimated_frequency']
114
+ print(f" {i}. {value} (frequency: {freq:,})")
115
+
116
+ print(f" 总进度: {self.completed_count}/{total} ({progress:.1f}%)")
117
+
118
+ return {
119
+ 'enhanced_values': enhanced_values,
120
+ 'reasoning': reasoning,
121
+ 'domain': final_domain, # 使用优化后的 domain name
122
+ 'original_domain': domain # 保留原始 domain name
123
+ }
124
+ else:
125
+ with self.lock:
126
+ self.completed_count += 1
127
+ print(f" ⚠️ LLM 响应缺少必需字段")
128
+ # 返回原始数据(格式统一)
129
+ return {
130
+ 'enhanced_values': [
131
+ {'value': v, 'estimated_frequency': c}
132
+ for v, c in values
133
+ ],
134
+ 'reasoning': 'LLM response format error, kept original',
135
+ 'domain': domain
136
+ }
137
+
138
+ else:
139
+ with self.lock:
140
+ self.completed_count += 1
141
+ print(f" ⚠️ LLM API 调用失败")
142
+ return {
143
+ 'enhanced_values': [
144
+ {'value': v, 'estimated_frequency': c}
145
+ for v, c in values
146
+ ],
147
+ 'reasoning': 'LLM API failed, kept original',
148
+ 'domain': domain
149
+ }
150
+
151
+ except json.JSONDecodeError as e:
152
+ with self.lock:
153
+ self.completed_count += 1
154
+ print(f" ⚠️ LLM 响应不是有效的 JSON: {e}")
155
+ print(f" 响应: {response[:300]}...")
156
+ return {
157
+ 'enhanced_values': [
158
+ {'value': v, 'estimated_frequency': c}
159
+ for v, c in values
160
+ ],
161
+ 'reasoning': 'JSON decode error, kept original',
162
+ 'domain': domain
163
+ }
164
+ except Exception as e:
165
+ with self.lock:
166
+ self.completed_count += 1
167
+ print(f" ⚠️ 增强错误: {e}")
168
+ return {
169
+ 'enhanced_values': [
170
+ {'value': v, 'estimated_frequency': c}
171
+ for v, c in values
172
+ ],
173
+ 'reasoning': f'Error: {str(e)}, kept original',
174
+ 'domain': domain
175
+ }
176
+
177
+ def _build_enhancement_prompt(self, domain: str, values: List[Tuple[str, int]]) -> str:
178
+ """构建增强的 prompt"""
179
+
180
+ # 构建现有 values 列表
181
+ values_str = '\n'.join([
182
+ f" - {value} (current count: {count})"
183
+ for value, count in values[:50] # 最多显示前50个
184
+ ])
185
+
186
+ if len(values) > 50:
187
+ values_str += f"\n ... and {len(values) - 50} more values"
188
+
189
+ prompt = f"""
190
+ You are a data curation and domain expert. Given a domain and its current list of values, your task is to produce a clean, comprehensive, and realistic set of values by following the steps below.
191
+
192
+ Your responsibilities:
193
+
194
+ 0. **REVIEW DOMAIN NAME**:
195
+ - First, evaluate whether the domain name accurately describes the values.
196
+ - If the domain name is ambiguous, too broad, too narrow, or doesn't properly represent the values, suggest a better, more accurate domain name.
197
+ - The refined domain name should clearly indicate what category/dimension these values represent.
198
+ - Example: "Category" → "Product Category"; "Type" → "Vehicle Type"; "Name" → "Brand Name"
199
+
200
+ 1. **KEEP** all existing values that are meaningful, commonly known, and relevant to the domain.
201
+ 2. **ADD** important missing values that are widely recognized in the real world for this domain.
202
+ 3. **REMOVE** values that are obscure, extremely niche, rarely used, outdated, or not commonly recognized.
203
+ 4. **DEDUPLICATE** values:
204
+ - Remove exact duplicates.
205
+ - Merge near-duplicates or synonyms (e.g., spelling variants, abbreviations, singular/plural forms).
206
+ - Normalize values so that each real-world concept appears **only once**.
207
+ 5. **REMOVE REDUNDANT ATTRIBUTES**:
208
+ - If values contain extra attributes (e.g., qualifiers, parenthetical notes, versions, descriptors),
209
+ keep only the **canonical, clean value name** unless the attribute is essential to distinguish meaning.
210
+ - Example: "Football (Soccer)" → "Soccer"; "iPhone 14 Pro Max" → "iPhone 14 Pro" (if variants are not required).
211
+ 6. **ESTIMATE** a real-world frequency/popularity score for each final value (scale: 1–10000),
212
+ reflecting how commonly the value is encountered or recognized in practice.
213
+ 7. **SORT** all final values by estimated real-world frequency, from most common/popular to least.
214
+
215
+ Domain:
216
+ "{domain}"
217
+
218
+ Current Values:
219
+ {values_str}
220
+
221
+ Guidelines:
222
+ - For well-defined enumerable domains (e.g., US States, Countries), include **all standard items**.
223
+ - For category-style domains (e.g., Sports, Genres, Product Types), focus on **mainstream, widely recognized** values.
224
+ - Prefer canonical names over aliases or variants.
225
+ - Base frequency estimates on real-world usage, awareness, or prevalence — not on the input list.
226
+ - Add missing values that a knowledgeable human would reasonably expect to see.
227
+ - Aim for **20–50 values** for most domains (more only when the domain is inherently exhaustive).
228
+
229
+ Output Requirements:
230
+ - Return your response in the following JSON format ONLY.
231
+ - Do NOT include markdown, comments, or explanatory text outside the JSON.
232
+ - Ensure the final list is fully deduplicated and normalized.
233
+ - If the domain name is appropriate, set "refined_domain_name" to null or the same name.
234
+ - If the domain name should be changed, provide a better, more specific name in "refined_domain_name".
235
+
236
+ {{
237
+ "refined_domain_name": "Better Domain Name" or null,
238
+ "enhanced_values": [
239
+ {{"value": "ValueName1", "estimated_frequency": 10000}},
240
+ {{"value": "ValueName2", "estimated_frequency": 8500}}
241
+ ],
242
+ "reasoning": "Brief explanation of domain name refinement (if any), key additions, removals, deduplication decisions, and sorting rationale."
243
+ }}
244
+
245
+ Examples:
246
+
247
+ Example — Domain: "US State"
248
+ - DOMAIN NAME: Appropriate, keep as is (refined_domain_name: null)
249
+ - KEEP: All valid US states already present
250
+ - ADD: Any missing states to complete the full set of 50
251
+ - REMOVE: None
252
+ - DEDUPLICATE: Merge variants like "CA" and "California" → "California"
253
+ - SORT: By population (California, Texas, Florida, New York...)
254
+
255
+ Example — Domain: "Sport Type"
256
+ - DOMAIN NAME: Appropriate, keep as is (refined_domain_name: null)
257
+ - KEEP: Soccer, Basketball, Baseball, Tennis
258
+ - ADD: Football, Cricket, Swimming (if missing and globally popular)
259
+ - REMOVE: Extremely obscure sports unless they are regionally mainstream
260
+ - DEDUPLICATE: Merge "Football" and "American Football" appropriately based on context
261
+ - SORT: By global popularity (Soccer, Basketball, Cricket, Tennis...)
262
+
263
+ Example — Domain: "Category" (with values like "Electronics", "Books", "Clothing")
264
+ - DOMAIN NAME: Too generic → refined_domain_name: "Product Category"
265
+ - Values processing continues as normal...
266
+
267
+ Example — Domain: "Type" (with values like "SUV", "Sedan", "Truck")
268
+ - DOMAIN NAME: Too generic → refined_domain_name: "Vehicle Type"
269
+ - Values processing continues as normal...
270
+ """
271
+
272
+ return prompt
273
+
274
+ def _query_llm(self, prompt: str) -> Optional[str]:
275
+ """查询 LLM API"""
276
+ headers = {
277
+ 'Authorization': f'Bearer {self.api_key}',
278
+ 'Content-Type': 'application/json'
279
+ }
280
+
281
+ data = {
282
+ 'model': self.model,
283
+ 'messages': [
284
+ {
285
+ 'role': 'system',
286
+ 'content': 'You are a data expert specialized in enhancing and organizing domain-specific values based on real-world knowledge. Always return valid JSON format only, without any markdown formatting or extra text.'
287
+ },
288
+ {
289
+ 'role': 'user',
290
+ 'content': prompt
291
+ }
292
+ ],
293
+ 'temperature': 0.3
294
+ }
295
+
296
+ try:
297
+ response = requests.post(
298
+ f'{self.base_url}/chat/completions',
299
+ headers=headers,
300
+ json=data,
301
+ timeout=120
302
+ )
303
+ response.raise_for_status()
304
+
305
+ result = response.json()
306
+ return result['choices'][0]['message']['content'].strip()
307
+
308
+ except requests.exceptions.Timeout:
309
+ print(" ❌ LLM API 超时")
310
+ return None
311
+ except requests.exceptions.HTTPError as e:
312
+ print(f" ❌ LLM API HTTP 错误: {e}")
313
+ return None
314
+ except requests.exceptions.RequestException as e:
315
+ print(f" ❌ LLM API 请求错误: {e}")
316
+ return None
317
+ except KeyError as e:
318
+ print(f" ❌ LLM API 响应格式错误: {e}")
319
+ return None
320
+
321
+
322
+ def parse_csv_line(line: str) -> Tuple[str, str, int]:
323
+ """解析 CSV 行"""
324
+ parts = []
325
+ current = []
326
+ in_quotes = False
327
+
328
+ for char in line:
329
+ if char == '"':
330
+ in_quotes = not in_quotes
331
+ elif char == ',' and not in_quotes:
332
+ parts.append(''.join(current).strip())
333
+ current = []
334
+ else:
335
+ current.append(char)
336
+ parts.append(''.join(current).strip())
337
+
338
+ if len(parts) >= 3:
339
+ domain = parts[0]
340
+ attribute = parts[1]
341
+ count = int(parts[2])
342
+ return domain, attribute, count
343
+ else:
344
+ return None, None, 0
345
+
346
+
347
+ def load_accepted_domains(file_path: str) -> List[str]:
348
+ """加载可接受的 domains"""
349
+ domains = []
350
+ with open(file_path, 'r', encoding='utf-8') as f:
351
+ for line in f:
352
+ line = line.strip()
353
+ if not line:
354
+ continue
355
+ domains.append(line)
356
+ return domains
357
+
358
+
359
+ def aggregate_by_domain(file_path: str) -> Dict[str, List[Tuple[str, int]]]:
360
+ """按 domain 聚合数据"""
361
+ domain_data = defaultdict(list)
362
+
363
+ with open(file_path, 'r', encoding='utf-8') as f:
364
+ for line in f:
365
+ line = line.strip()
366
+ if not line:
367
+ continue
368
+
369
+ domain, attribute, count = parse_csv_line(line)
370
+ if domain:
371
+ domain_data[domain].append((attribute, count))
372
+
373
+ return domain_data
374
+
375
+
376
+ def main():
377
+ print("=" * 80)
378
+ print("Domain Values 增强和排序工具")
379
+ print("=" * 80)
380
+ print()
381
+
382
+ # 1. 加载可接受的 domains
383
+ print("📖 读取可接受的 domains...")
384
+ accepted_domains = load_accepted_domains('accepted_domains.txt')
385
+ print(f"✅ 可接受的 domains: {len(accepted_domains)} 个")
386
+ if len(accepted_domains) <= 20:
387
+ for idx, domain in enumerate(accepted_domains, 1):
388
+ print(f" {idx}. {domain}")
389
+ else:
390
+ print(f" 前 10 个: {', '.join(accepted_domains[:10])}")
391
+ print(f" ... 还有 {len(accepted_domains) - 10} 个")
392
+ print()
393
+
394
+ # 2. 加载和聚合数据
395
+ print("📖 读取和聚合数据...")
396
+ domain_data = aggregate_by_domain('domain_value_pairs_filtered.txt')
397
+ print(f"✅ 总共 {len(domain_data)} 个不同的 domains")
398
+ print()
399
+
400
+ # 3. 过滤只保留接受的 domains
401
+ filtered_data = {
402
+ domain: values
403
+ for domain, values in domain_data.items()
404
+ if domain in accepted_domains
405
+ }
406
+
407
+ print(f"🔍 过滤后保留 {len(filtered_data)} 个 domains")
408
+ if len(filtered_data) <= 20:
409
+ for domain in filtered_data.keys():
410
+ value_count = len(filtered_data[domain])
411
+ total_count = sum(c for _, c in filtered_data[domain])
412
+ print(f" - {domain}: {value_count} values, total count: {total_count:,}")
413
+ else:
414
+ print(" 前 10 个 domains:")
415
+ for i, domain in enumerate(list(filtered_data.keys())[:10], 1):
416
+ value_count = len(filtered_data[domain])
417
+ total_count = sum(c for _, c in filtered_data[domain])
418
+ print(f" {i}. {domain}: {value_count} values, total count: {total_count:,}")
419
+ print(f" ... 还有 {len(filtered_data) - 10} 个 domains")
420
+ print()
421
+
422
+ # 4. 初始化增强器
423
+ enhancer = ValueEnhancer()
424
+ enhancer.total_domains = len(filtered_data)
425
+
426
+ # 5. 准备任务列表
427
+ print("=" * 80)
428
+ print("🚀 开始处理和增强 domains (10线程并发)")
429
+ print("=" * 80)
430
+ print()
431
+
432
+ # 转换为列表以便索引
433
+ domain_items = list(filtered_data.items())
434
+ total_domains = len(domain_items)
435
+
436
+ enhanced_results = {}
437
+ start_time = time.time()
438
+
439
+ # 6. 使用线程池并发处理
440
+ num_threads = 15
441
+ print(f"🔧 使用 {num_threads} 个线程并发处理")
442
+ print()
443
+
444
+ with ThreadPoolExecutor(max_workers=num_threads) as executor:
445
+ # 提交所有任务
446
+ future_to_domain = {
447
+ executor.submit(
448
+ enhancer.enhance_and_sort_values,
449
+ domain,
450
+ values,
451
+ idx,
452
+ total_domains,
453
+ start_time
454
+ ): domain
455
+ for idx, (domain, values) in enumerate(domain_items, 1)
456
+ }
457
+
458
+ # 收集结果
459
+ domain_name_mapping = {} # 原始 domain -> 优化后 domain 的映射
460
+
461
+ for future in as_completed(future_to_domain):
462
+ original_domain = future_to_domain[future]
463
+ try:
464
+ result = future.result()
465
+ final_domain = result['domain']
466
+ enhanced_results[final_domain] = result['enhanced_values']
467
+
468
+ # 记录 domain name 映射
469
+ if final_domain != original_domain:
470
+ domain_name_mapping[original_domain] = final_domain
471
+
472
+ except Exception as e:
473
+ print(f"❌ 处理 {original_domain} 时出错: {e}")
474
+ # 保留原始数据
475
+ enhanced_results[original_domain] = [
476
+ {'value': v, 'estimated_frequency': c}
477
+ for v, c in filtered_data[original_domain]
478
+ ]
479
+
480
+ # 6. 保存结果
481
+ elapsed_time = time.time() - start_time
482
+
483
+ print()
484
+ print("=" * 80)
485
+ print("💾 保存结果...")
486
+ print("=" * 80)
487
+
488
+ # 显示 domain name 变化(如果有)
489
+ if domain_name_mapping:
490
+ print()
491
+ print(f"🔄 Domain 名称优化记录 ({len(domain_name_mapping)} 个):")
492
+ for original, refined in domain_name_mapping.items():
493
+ print(f" {original} → {refined}")
494
+ print()
495
+
496
+ output_file = 'domain_value_pairs_enhanced.txt'
497
+
498
+ # 展开所有 domain-value pairs 并按 frequency 全局排序
499
+ all_pairs = []
500
+ for domain, values in enhanced_results.items():
501
+ for item in values:
502
+ value = item['value']
503
+ freq = item['estimated_frequency']
504
+ all_pairs.append((domain, value, freq))
505
+
506
+ # 按 frequency 降序排序
507
+ all_pairs.sort(key=lambda x: x[2], reverse=True)
508
+
509
+ # 保存到文件
510
+ with open(output_file, 'w', encoding='utf-8') as f:
511
+ for domain, value, freq in all_pairs:
512
+ # 处理可能包含逗号的字段
513
+ if ',' in value:
514
+ value = f'"{value}"'
515
+ if ',' in domain:
516
+ domain = f'"{domain}"'
517
+ f.write(f"{domain},{value},{freq}\n")
518
+
519
+ print(f"✅ 结果已保存到: {output_file}")
520
+ print()
521
+
522
+ # 7. 统计信息
523
+ print("=" * 80)
524
+ print("📊 统计信息:")
525
+ print("=" * 80)
526
+ print(f"处理的 domains: {len(enhanced_results)}")
527
+ print(f"总 pairs 数: {len(all_pairs)}")
528
+ print(f"总耗时: {elapsed_time:.1f} 秒")
529
+ print()
530
+
531
+ # 按 domain 显示统计
532
+ print("各 Domain 统计:")
533
+ for domain in accepted_domains:
534
+ if domain in enhanced_results:
535
+ count = len(enhanced_results[domain])
536
+ print(f" {domain}: {count} values")
537
+ print()
538
+
539
+ # 显示 Top 20 pairs
540
+ print("🏆 Top 20 Pairs (按 estimated frequency):")
541
+ for i, (domain, value, freq) in enumerate(all_pairs[:20], 1):
542
+ print(f" {i:2d}. {domain}: {value} (frequency: {freq:,})")
543
+
544
+ print()
545
+ print("=" * 80)
546
+ print("✨ 增强完成!")
547
+ print("=" * 80)
548
+
549
+
550
+ if __name__ == '__main__':
551
+ main()
icon_generation/backup/enhance_log.txt ADDED
@@ -0,0 +1,1130 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ================================================================================
2
+ Domain Values 增强和排序工具
3
+ ================================================================================
4
+
5
+ 📖 读取可接受的 domains...
6
+ ✅ 可接受的 domains: 609 个
7
+ 1. US State
8
+ 2. Entertainment Genre
9
+ 3. Sport Type
10
+ 4. Digital Platform
11
+ 5. Industry Sector
12
+ 6. Gender Demographic
13
+ 7. Product Category
14
+ 8. Canadian Province
15
+ 9. Crop Type
16
+ 10. Geographic Classification
17
+ 11. Sports Team
18
+ 12. Animal Species
19
+ 13. Browser Type
20
+ 14. Device Type
21
+ 15. School Category
22
+ 16. Art Medium
23
+ 17. Material
24
+ 18. Specialization Area
25
+ 19. Energy Technology
26
+ 20. Trade Direction
27
+ 21. Dog Breed
28
+ 22. Continent
29
+ 23. Energy Source
30
+ 24. Occurrence Frequency
31
+ 25. Economic System
32
+ 26. Country
33
+ 27. Transportation Mode
34
+ 28. Travel Location
35
+ 29. Company Name
36
+ 30. Income Bracket
37
+ 31. Sports League
38
+ 32. Smart Home Category
39
+ 33. Media Type
40
+ 34. Academic Subject
41
+ 35. Land Use Type
42
+ 36. Vehicle Classification
43
+ 37. Irrigation Method
44
+ 38. Calendar Month
45
+ 39. Socioeconomic Status Level
46
+ 40. Farm Size Category
47
+ 41. Digital Service Provider
48
+ 42. Political Party
49
+ 43. TV Network
50
+ 44. Physical Activity
51
+ 45. Dance Discipline
52
+ 46. Musical Instrument
53
+ 47. Gemstone Type
54
+ 48. Building Classification
55
+ 49. Product Brand
56
+ 50. Monetization Method
57
+ 51. Airline Name
58
+ 52. Clothing Type
59
+ 53. Football Club
60
+ 54. Cuisine
61
+ 55. Beer Style
62
+ 56. Diplomatic Mission Type
63
+ 57. Political Alignment
64
+ 58. Airport Code
65
+ 59. Operation Type
66
+ 60. Home Appliance
67
+ 61. Seaport
68
+ 62. Client Engagement Channel
69
+ 63. Attraction Park Name
70
+ 64. Farming System
71
+ 65. Superhero
72
+ 66. Disease Name
73
+ 67. Occupational Role
74
+ 68. Organization Type
75
+ 69. Infrastructure Type
76
+ 70. Athletic Conference
77
+ 71. Integrity Standing
78
+ 72. Toy Type
79
+ 73. Creative Artist
80
+ 74. Administrative Borough
81
+ 75. Biome Type
82
+ 76. Event Classification
83
+ 77. Seafood Species
84
+ 78. Tea Type
85
+ 79. Leisure Activity
86
+ 80. Expense Type
87
+ 81. Smartphone Brand
88
+ 82. Landmark Name
89
+ 83. Audience Segment
90
+ 84. Occupation
91
+ 85. Calendar Season
92
+ 86. Subject Area
93
+ 87. Incident Type
94
+ 88. Tillage System
95
+ 89. Protected Area Name
96
+ 90. Art Technique
97
+ 91. Fabric Type
98
+ 92. Extreme Temperature Type
99
+ 93. Island Name
100
+ 94. Podcast Genre
101
+ 95. Time Off Type
102
+ 96. Pet Category
103
+ 97. Medal Type
104
+ 98. Opinion Type
105
+ 99. Playground Feature
106
+ 100. Video Game Title
107
+ 101. Regulatory Domain
108
+ 102. Streaming Service
109
+ 103. Vehicle Model
110
+ 104. Game Name
111
+ 105. Demographic Group
112
+ 106. Cooking Technique
113
+ 107. Media Asset Type
114
+ 108. Bird Species
115
+ 109. Religious Affiliation
116
+ 110. Exhibit Theme
117
+ 111. Medical Specialty
118
+ 112. Weather Type
119
+ 113. Artifact Class
120
+ 114. Brewing Method
121
+ 115. Dwelling Type
122
+ 116. Company Size Category
123
+ 117. City
124
+ 118. Color
125
+ 119. Family Structure Type
126
+ 120. Performance Type
127
+ 121. Craft Category
128
+ 122. Nutritional Component
129
+ 123. Music Genre
130
+ 124. Venue Type
131
+ 125. Marine Conservation Name
132
+ 126. Pollution Cause Type
133
+ 127. Medical Procedure
134
+ 128. Property Type
135
+ 129. Radio Genre
136
+ 130. Art Period
137
+ 131. Phone Model
138
+ 132. Business Segment
139
+ 133. Soil Texture
140
+ 134. Central Bank Name
141
+ 135. Cover Crop Type
142
+ 136. Road Classification
143
+ 137. Dietary Consideration
144
+ 138. Beverage Type
145
+ 139. Vehicle Powertrain
146
+ 140. Lake Name
147
+ 141. River Basin Name
148
+ 142. Accessory Category
149
+ 143. Jewelry Design Style
150
+ 144. Agricultural Practice
151
+ 145. Ecosystem Type
152
+ 146. Home Space Type
153
+ 147. Marital Status
154
+ 148. Building Style
155
+ 149. Day Classification
156
+ 150. Produce Name
157
+ 151. Medical Injury Type
158
+ 152. Risk Rating
159
+ 153. Payment Method
160
+ 154. Education Category
161
+ 155. Proficiency Level
162
+ 156. Investment Asset Class
163
+ 157. Park Category
164
+ 158. Purchase Channel
165
+ 159. Protein Type
166
+ 160. Class Classification
167
+ 161. Casualty Status
168
+ 162. Phone Type
169
+ 163. Camera Form Factor
170
+ 164. Travel Category
171
+ 165. Recipe Ingredient
172
+ 166. Flight Disruption Type
173
+ 167. Musical Title
174
+ 168. Food Category
175
+ 169. Generational Role
176
+ 170. National Park Name
177
+ 171. Show Title
178
+ 172. Interest Category
179
+ 173. Supply Chain Stage
180
+ 174. Comedy Type
181
+ 175. Family Member Count
182
+ 176. Metal Type
183
+ 177. Historical Period
184
+ 178. Day Period
185
+ 179. Manufacturing Stage
186
+ 180. Retail Environment
187
+ 181. Crypto Exchange
188
+ 182. Customer Segment Name
189
+ 183. Holiday Name
190
+ 184. Recipient Category
191
+ 185. Minority Indicator
192
+ 186. Medical Condition
193
+ 187. Louisiana Parish
194
+ 188. Performance Indicator
195
+ 189. Sports Facility Type
196
+ 190. Pest Category
197
+ 191. Propulsion Type
198
+ 192. Nuclear Reactor Type
199
+ 193. Game Genre
200
+ 194. Revenue Type
201
+ 195. Financial Product Type
202
+ 196. Galaxy Type
203
+ 197. Environmental Impact
204
+ 198. Craft Supply
205
+ 199. Temperature Record Type
206
+ 200. Furniture Category
207
+ 201. Fruit Name
208
+ 202. Social Event Type
209
+ 203. Neighborhood
210
+ 204. Service Category
211
+ 205. Zoning District
212
+ 206. Play Type
213
+ 207. Wine Classification
214
+ 208. Eating Occasion
215
+ 209. Operating System
216
+ 210. Career Level
217
+ 211. Music Subgenre
218
+ 212. Aircraft Model
219
+ 213. Major Sports Event
220
+ 214. Disaster Event
221
+ 215. Harvest Method
222
+ 216. Marine Area Name
223
+ 217. Involvement Level
224
+ 218. Gemstone Color
225
+ 219. Tourist Attraction
226
+ 220. Fantasy Entity
227
+ 221. Household Life Stage
228
+ 222. Climate Type
229
+ 223. Air Quality Parameter
230
+ 224. News Organization
231
+ 225. Heritage Site Type
232
+ 226. Earring Type
233
+ 227. Fertilizer Type
234
+ 228. Health Area
235
+ 229. Conservation Status
236
+ 230. Education Format
237
+ 231. System Component
238
+ 232. Gaming Publication
239
+ 233. Animal Type
240
+ 234. Animal Adoption Source
241
+ 235. Capture Method
242
+ 236. Festival Theme
243
+ 237. Participation Level
244
+ 238. Climate Zone Type
245
+ 239. Achievement Award Category
246
+ 240. Capital City
247
+ 241. Waste Material
248
+ 242. Outreach Channel
249
+ 243. Reef Site
250
+ 244. Exercise Type
251
+ 245. Cancer Type
252
+ 246. Content Streaming Platform
253
+ 247. Turf Type
254
+ 248. Home Decor Category
255
+ 249. Geographic Region
256
+ 250. Establishment Type
257
+ 251. Mitigation Method
258
+ 252. Grant Active
259
+ 253. Storage Category
260
+ 254. Concession Product
261
+ 255. Citrus Variety
262
+ 256. Coffee Style
263
+ 257. Feedstock Type
264
+ 258. Insurance Type
265
+ 259. Land Management Practice
266
+ 260. Application Category
267
+ 261. Lighting Fixture Type
268
+ 262. Data Domain
269
+ 263. Award Ceremony
270
+ 264. Conservation Organization
271
+ 265. Music Distribution Format
272
+ 266. Social Media Intensity
273
+ 267. Engagement
274
+ 268. Primate Species
275
+ 269. Grain Category
276
+ 270. Tourism Type
277
+ 271. Launch Status
278
+ 272. Funding Source
279
+ 273. Email Provider
280
+ 274. Amusement Ride Type
281
+ 275. Vessel Classification
282
+ 276. Fatal Incident Type
283
+ 277. Mission Complexity Level
284
+ 278. Mobile OS
285
+ 279. Museum Focus
286
+ 280. Market Trend
287
+ 281. Enforcement Degree
288
+ 282. Flyway Region
289
+ 283. Travel Segment
290
+ 284. Usage Scenario
291
+ 285. Membership Category
292
+ 286. Weightlifting Exercise
293
+ 287. Lighting Type
294
+ 288. Fitness Goal
295
+ 289. Vineyard Designation
296
+ 290. Food Product Name
297
+ 291. Political Entity
298
+ 292. Economic Region
299
+ 293. Health Coverage Type
300
+ 294. Data Collection Method
301
+ 295. Innovation Area
302
+ 296. Food Item
303
+ 297. Drug Name
304
+ 298. Store Category
305
+ 299. Wearable Item Type
306
+ 300. Photography Genre
307
+ 301. Tomato Variety
308
+ 302. Therapeutic Area
309
+ 303. Sport Position
310
+ 304. Packaging Material
311
+ 305. Fashion Style
312
+ 306. Cargo Type
313
+ 307. Sensor Category
314
+ 308. Public Policy
315
+ 309. Military Platform Type
316
+ 310. Plant Establishment Method
317
+ 311. Marketing Channel
318
+ 312. Education Level
319
+ 313. Project Type
320
+ 314. Listening Method
321
+ 315. Social Determinant
322
+ 316. Art Movement
323
+ 317. Higher Education Type
324
+ 318. Agricultural Type
325
+ 319. Storage Type
326
+ 320. Measure Type
327
+ 321. ADAS Module
328
+ 322. Debris Material
329
+ 323. Retailer Name
330
+ 324. Legislative Status
331
+ 325. Engineering Structure Type
332
+ 326. Pesticide Classification
333
+ 327. Program Certification Level
334
+ 328. Pandemic Period
335
+ 329. Financial Incentive Type
336
+ 330. Dish Type
337
+ 331. Impact Area
338
+ 332. Shipping Passage
339
+ 333. Proximity Level
340
+ 334. Sofa Type
341
+ 335. Content Section
342
+ 336. Corridor Name
343
+ 337. Fixture Design
344
+ 338. Immigration Status
345
+ 339. Project Category
346
+ 340. Painting Medium
347
+ 341. Gift Category
348
+ 342. Sanction Measure
349
+ 343. Forest Cover Type
350
+ 344. Software Application
351
+ 345. Currency Code
352
+ 346. Insulation Material
353
+ 347. Aid Sector
354
+ 348. Health Condition Category
355
+ 349. Art Subject
356
+ 350. Vehicle Maintenance Type
357
+ 351. Tractor Configuration
358
+ 352. Rover Name
359
+ 353. Product Variant
360
+ 354. Student Proficiency
361
+ 355. Costume Character
362
+ 356. Event Category
363
+ 357. Seabird Species
364
+ 358. Service Branch
365
+ 359. Flag State Name
366
+ 360. Insect Common Name
367
+ 361. Accessibility Feature Type
368
+ 362. Camera Lens Type
369
+ 363. Roofing Type
370
+ 364. Power Facility
371
+ 365. Comic Issue
372
+ 366. Sustainability Category
373
+ 367. Court Surface
374
+ 368. Consumer Profile
375
+ 369. Technology Solution
376
+ 370. Historic Site Name
377
+ 371. Accommodation Name
378
+ 372. Workout Modality
379
+ 373. Season Phase
380
+ 374. Satellite Function
381
+ 375. Sales Offer Type
382
+ 376. Driving Context
383
+ 377. Professional Skill
384
+ 378. GDP Component Type
385
+ 379. App Pricing Model
386
+ 380. Cultural Attire
387
+ 381. Planet Name
388
+ 382. Dietary Food Group
389
+ 383. Technique Type
390
+ 384. Wellbeing Dimension
391
+ 385. Livestock Category
392
+ 386. Care Level
393
+ 387. Campaign Type
394
+ 388. Shot Zone
395
+ 389. Acquisition Channel
396
+ 390. Work Gear Type
397
+ 391. Light Pollution Severity
398
+ 392. Source Collection
399
+ 393. App Permission
400
+ 394. Religious Symbol
401
+ 395. Manufacturing Step
402
+ 396. App Usage Level
403
+ 397. Disruption Reason
404
+ 398. Funding Type
405
+ 399. Sustainable Feature
406
+ 400. Green Space Accessibility
407
+ 401. Fishery Location
408
+ 402. Workshop Discipline
409
+ 403. Developmental Skill Category
410
+ 404. Cryptocurrency Name
411
+ 405. Adoption Level
412
+ 406. Hurricane Wind Category
413
+ 407. Production Type
414
+ 408. Pollution Source
415
+ 409. Support Category
416
+ 410. Amphibian Species Name
417
+ 411. Sales Metric
418
+ 412. Alcoholic Beverage Type
419
+ 413. Property Flooring Type
420
+ 414. Resource Classification
421
+ 415. Orchard Region
422
+ 416. Ballpark Name
423
+ 417. Home Feature
424
+ 418. Flight Route
425
+ 419. Battleground State
426
+ 420. Parental Involvement Level
427
+ 421. Interaction Channel
428
+ 422. Booking Point
429
+ 423. Instructional Method
430
+ 424. Space Agency
431
+ 425. Polymer Material
432
+ 426. Performance Standing
433
+ 427. Connected Service
434
+ 428. Cyber Attack Type
435
+ 429. Irrigation Source Type
436
+ 430. Organization Sector
437
+ 431. Ballet Title
438
+ 432. Hardware Category
439
+ 433. CNY Food
440
+ 434. Vehicle Part Type
441
+ 435. Site Name
442
+ 436. Price Level
443
+ 437. Manufacturing Process
444
+ 438. Participation Status
445
+ 439. Affectionate Gesture
446
+ 440. IP Infringement Type
447
+ 441. Mitigation Project Type
448
+ 442. Flavor Profile
449
+ 443. Research Publication Type
450
+ 444. Footwear Style
451
+ 445. Environmental Strategy
452
+ 446. Tributary Name
453
+ 447. Cultivation Environment
454
+ 448. Transit Station
455
+ 449. Microphone Classification
456
+ 450. Economic Indicator Type
457
+ 451. Remediation Method
458
+ 452. Project Phase
459
+ 453. Financial Concern Category
460
+ 454. Manure Management Method
461
+ 455. Set Category
462
+ 456. Education Focus
463
+ 457. Permit Sector
464
+ 458. Body Shape
465
+ 459. Building Siding Material
466
+ 460. Sport Tier
467
+ 461. Filtration Technology
468
+ 462. Theme Park Name
469
+ 463. Manufacturing Machine Type
470
+ 464. Space Mission
471
+ 465. Agricultural Region
472
+ 466. Essential Need
473
+ 467. Craft Material
474
+ 468. Management Technique
475
+ 469. Car Model Name
476
+ 470. Zone Designation
477
+ 471. Design Style
478
+ 472. Pruning Style
479
+ 473. Sport Gender Division
480
+ 474. Water Supply Type
481
+ 475. Mammal Species Name
482
+ 476. Insect Type
483
+ 477. Product Form
484
+ 478. Control Technique
485
+ 479. Clinical Service
486
+ 480. Component Type
487
+ 481. Probability Basis
488
+ 482. Hub City
489
+ 483. Internet Access Status
490
+ 484. Tariff Active
491
+ 485. Hat Style
492
+ 486. Pet Service Type
493
+ 487. Computer Peripheral
494
+ 488. Farming Diversification Strategy
495
+ 489. Cost Element
496
+ 490. Impact Severity
497
+ 491. Washing Machine Form Factor
498
+ 492. Spending Occasion
499
+ 493. Video Platform
500
+ 494. Election Type
501
+ 495. Milestone Type
502
+ 496. Index Basis
503
+ 497. Loan Application Status
504
+ 498. Farm Activity
505
+ 499. Scenario Type
506
+ 500. Gallery Sector
507
+ 501. Forage Type
508
+ 502. Ecological Area Type
509
+ 503. Application Name
510
+ 504. Patrol Method
511
+ 505. Access Level
512
+ 506. Result Status
513
+ 507. Subsidy Indicator
514
+ 508. Subscription Plan
515
+ 509. Light Color
516
+ 510. Wellness Program Type
517
+ 511. Educational Attainment
518
+ 512. Agricultural Pest
519
+ 513. Library Name
520
+ 514. Subject Category
521
+ 515. Reef Region
522
+ 516. Medicine System
523
+ 517. National Leader
524
+ 518. Observatory Name
525
+ 519. Budget Category
526
+ 520. Flood Risk Level
527
+ 521. Pet Package Type
528
+ 522. Voter Segment
529
+ 523. Motorcycle Type
530
+ 524. Message Purpose
531
+ 525. Conflict Country
532
+ 526. Cultural Institution Type
533
+ 527. Mission Operation
534
+ 528. Narrative Device
535
+ 529. Japan Prefecture
536
+ 530. Circuit Type
537
+ 531. Cultural Offering
538
+ 532. Character Kind
539
+ 533. Medical Intervention
540
+ 534. Hazard Classification
541
+ 535. Shoe Segment
542
+ 536. Software Function
543
+ 537. Tour Activity
544
+ 538. User Segment
545
+ 539. Pricing Basis
546
+ 540. Culinary Trend
547
+ 541. Trade Barrier Type
548
+ 542. Dress Silhouette
549
+ 543. Performance Terrain
550
+ 544. Aircraft Category
551
+ 545. Competitor Brand
552
+ 546. Discovery Category
553
+ 547. Personal Data Type
554
+ 548. Beneficiary Type
555
+ 549. Program Type
556
+ 550. Theater Format
557
+ 551. Love Language Type
558
+ 552. Pollinator Classification
559
+ 553. Magazine Name
560
+ 554. Patient Risk Group
561
+ 555. Healthcare Facility Type
562
+ 556. Endorsement Category
563
+ 557. Subscription Action Type
564
+ 558. Geographic Coast
565
+ 559. Venue Seating Section
566
+ 560. Worker Type
567
+ 561. Route Type
568
+ 562. Passenger Persona
569
+ 563. Employment Model
570
+ 564. Gift Type
571
+ 565. Taxon Name
572
+ 566. Therapy Modality
573
+ 567. Driving Scenario
574
+ 568. Blockchain Platform
575
+ 569. Federal Agency
576
+ 570. Environmental Regulation
577
+ 571. Ranching Approach
578
+ 572. Personal Relationship
579
+ 573. Sourcing Material Type
580
+ 574. Comic Series
581
+ 575. Development Type
582
+ 576. Vocational Trade
583
+ 577. Artwork Title
584
+ 578. Folklore Form
585
+ 579. Ritual Type
586
+ 580. Device Capability
587
+ 581. Treatment Type
588
+ 582. Ad Format
589
+ 583. Maize Cultivar
590
+ 584. Sacred Site Name
591
+ 585. Leisure Amenity
592
+ 586. Student Type
593
+ 587. Loyalty Program Feature
594
+ 588. Art Installation Type
595
+ 589. Plantation Company
596
+ 590. Damage Nature
597
+ 591. Server Role
598
+ 592. Subsidy Program Name
599
+ 593. RPG System
600
+ 594. Emotion Type
601
+ 595. App Function
602
+ 596. Freight Mode
603
+ 597. Customer Tier
604
+ 598. Art Tradition
605
+ 599. Action Type
606
+ 600. Market Segment
607
+ 601. Narrative Trope
608
+ 602. Home Improvement Type
609
+ 603. Debt Presence
610
+ 604. Ticket Category
611
+ 605. User Generation
612
+ 606. Lifecycle Status
613
+ 607. Playoff Round
614
+ 608. Donor Classification
615
+ 609. Collision Type
616
+
617
+ 📖 读取和聚合数据...
618
+ ✅ 总共 1099 个不同的 domains
619
+
620
+ 🔍 过滤后保留 609 个 domains:
621
+ - US State: 50 values, total count: 57,803
622
+ - Gender Demographic: 2 values, total count: 11,466
623
+ - Sport Type: 52 values, total count: 23,756
624
+ - Entertainment Genre: 19 values, total count: 24,732
625
+ - Digital Platform: 72 values, total count: 17,662
626
+ - Trade Direction: 2 values, total count: 4,266
627
+ - School Category: 6 values, total count: 4,716
628
+ - Economic System: 2 values, total count: 4,034
629
+ - Occurrence Frequency: 3 values, total count: 4,035
630
+ - Canadian Province: 10 values, total count: 9,192
631
+ - Industry Sector: 136 values, total count: 11,813
632
+ - Smart Home Category: 5 values, total count: 2,786
633
+ - Income Bracket: 3 values, total count: 3,501
634
+ - Monetization Method: 5 values, total count: 1,578
635
+ - Browser Type: 6 values, total count: 6,242
636
+ - Media Type: 15 values, total count: 2,681
637
+ - Product Category: 236 values, total count: 10,987
638
+ - Device Type: 31 values, total count: 5,665
639
+ - Transportation Mode: 20 values, total count: 3,910
640
+ - Art Medium: 25 values, total count: 4,653
641
+ - Crop Type: 84 values, total count: 8,647
642
+ - Continent: 7 values, total count: 4,206
643
+ - Land Use Type: 34 values, total count: 2,488
644
+ - Sports League: 6 values, total count: 2,962
645
+ - Time Off Type: 1 values, total count: 752
646
+ - Political Party: 10 values, total count: 1,961
647
+ - Socioeconomic Status Level: 3 values, total count: 2,098
648
+ - Farm Size Category: 3 values, total count: 2,082
649
+ - Specialization Area: 65 values, total count: 4,396
650
+ - Calendar Month: 12 values, total count: 2,119
651
+ - Smartphone Brand: 8 values, total count: 862
652
+ - Operation Type: 3 values, total count: 1,290
653
+ - Energy Technology: 73 values, total count: 4,387
654
+ - Vehicle Classification: 20 values, total count: 2,411
655
+ - Seaport: 5 values, total count: 1,216
656
+ - Dog Breed: 25 values, total count: 4,251
657
+ - Irrigation Method: 16 values, total count: 2,168
658
+ - Farming System: 4 values, total count: 1,148
659
+ - TV Network: 7 values, total count: 1,841
660
+ - Energy Source: 26 values, total count: 4,164
661
+ - Geographic Classification: 253 values, total count: 7,141
662
+ - Diplomatic Mission Type: 3 values, total count: 1,386
663
+ - Material: 75 values, total count: 4,498
664
+ - Extreme Temperature Type: 4 values, total count: 772
665
+ - Political Alignment: 7 values, total count: 1,379
666
+ - Academic Subject: 41 values, total count: 2,599
667
+ - Pet Category: 3 values, total count: 750
668
+ - Infrastructure Type: 5 values, total count: 1,015
669
+ - Product Brand: 38 values, total count: 1,614
670
+ - Client Engagement Channel: 8 values, total count: 1,197
671
+ - Physical Activity: 25 values, total count: 1,838
672
+ - Airport Code: 11 values, total count: 1,335
673
+ - Country: 41 values, total count: 4,019
674
+ - Dance Discipline: 16 values, total count: 1,821
675
+ - Beer Style: 14 values, total count: 1,390
676
+ - Regulatory Domain: 16 values, total count: 693
677
+ - Sports Team: 199 values, total count: 6,877
678
+ - Gemstone Type: 16 values, total count: 1,703
679
+ - Airline Name: 17 values, total count: 1,521
680
+ - Audience Segment: 21 values, total count: 845
681
+ - Clothing Type: 16 values, total count: 1,517
682
+ - Tea Type: 10 values, total count: 892
683
+ - Musical Instrument: 32 values, total count: 1,730
684
+ - Incident Type: 8 values, total count: 842
685
+ - Building Classification: 19 values, total count: 1,693
686
+ - Home Appliance: 11 values, total count: 1,235
687
+ - Medal Type: 3 values, total count: 750
688
+ - Company Size Category: 3 values, total count: 523
689
+ - Occupation: 13 values, total count: 845
690
+ - Soil Texture: 2 values, total count: 414
691
+ - Calendar Season: 4 values, total count: 844
692
+ - Company Name: 145 values, total count: 3,551
693
+ - Athletic Conference: 8 values, total count: 1,001
694
+ - Demographic Group: 10 values, total count: 613
695
+ - Cuisine: 21 values, total count: 1,450
696
+ - Media Asset Type: 7 values, total count: 607
697
+ - Musical Title: 4 values, total count: 302
698
+ - Administrative Borough: 5 values, total count: 932
699
+ - Fabric Type: 8 values, total count: 783
700
+ - Comedy Type: 3 values, total count: 269
701
+ - Animal Species: 247 values, total count: 6,478
702
+ - Performance Type: 6 values, total count: 482
703
+ - Travel Location: 70 values, total count: 3,717
704
+ - Phone Type: 2 values, total count: 308
705
+ - Event Classification: 14 values, total count: 922
706
+ - Art Period: 6 values, total count: 418
707
+ - Vehicle Powertrain: 3 values, total count: 386
708
+ - Digital Service Provider: 40 values, total count: 1,980
709
+ - Property Type: 10 values, total count: 422
710
+ - Tillage System: 9 values, total count: 824
711
+ - Integrity Standing: 11 values, total count: 996
712
+ - Proficiency Level: 3 values, total count: 329
713
+ - Opinion Type: 5 values, total count: 750
714
+ - Superhero: 15 values, total count: 1,075
715
+ - Streaming Service: 9 values, total count: 689
716
+ - Operating System: 3 values, total count: 204
717
+ - Religious Affiliation: 10 values, total count: 599
718
+ - Weather Type: 9 values, total count: 587
719
+ - Cooking Technique: 7 values, total count: 610
720
+ - Flight Disruption Type: 4 values, total count: 304
721
+ - Casualty Status: 2 values, total count: 309
722
+ - Subject Area: 14 values, total count: 843
723
+ - Household Life Stage: 2 values, total count: 189
724
+ - Animal Adoption Source: 2 values, total count: 171
725
+ - Disease Name: 15 values, total count: 1,049
726
+ - Podcast Genre: 10 values, total count: 761
727
+ - Grant Active: 1 values, total count: 151
728
+ - Storage Category: 1 values, total count: 151
729
+ - Family Structure Type: 6 values, total count: 493
730
+ - Exhibit Theme: 16 values, total count: 596
731
+ - Biome Type: 14 values, total count: 925
732
+ - Organization Type: 27 values, total count: 1,020
733
+ - Football Club: 27 values, total count: 1,472
734
+ - Craft Category: 9 values, total count: 461
735
+ - Retail Environment: 5 values, total count: 257
736
+ - Temperature Record Type: 2 values, total count: 220
737
+ - Marital Status: 8 values, total count: 348
738
+ - Playground Feature: 11 values, total count: 722
739
+ - Creative Artist: 42 values, total count: 936
740
+ - Nuclear Reactor Type: 3 values, total count: 228
741
+ - Central Bank Name: 4 values, total count: 404
742
+ - Art Technique: 20 values, total count: 790
743
+ - Road Classification: 9 values, total count: 401
744
+ - Manufacturing Stage: 2 values, total count: 259
745
+ - Nutritional Component: 9 values, total count: 461
746
+ - Park Category: 5 values, total count: 326
747
+ - Island Name: 22 values, total count: 767
748
+ - Bird Species: 20 values, total count: 600
749
+ - Minority Indicator: 2 values, total count: 246
750
+ - Camera Form Factor: 6 values, total count: 308
751
+ - Radio Genre: 7 values, total count: 420
752
+ - Risk Rating: 3 values, total count: 331
753
+ - Venue Type: 7 values, total count: 451
754
+ - Artifact Class: 14 values, total count: 584
755
+ - Leisure Activity: 23 values, total count: 877
756
+ - Toy Type: 23 values, total count: 956
757
+ - Day Classification: 6 values, total count: 344
758
+ - Business Segment: 6 values, total count: 415
759
+ - Pollution Cause Type: 10 values, total count: 434
760
+ - Home Space Type: 10 values, total count: 364
761
+ - Attraction Park Name: 49 values, total count: 1,169
762
+ - Brewing Method: 10 values, total count: 541
763
+ - Dietary Consideration: 6 values, total count: 399
764
+ - Proximity Level: 1 values, total count: 110
765
+ - Medical Injury Type: 5 values, total count: 335
766
+ - Immigration Status: 1 values, total count: 108
767
+ - Protein Type: 6 values, total count: 313
768
+ - Political Entity: 3 values, total count: 133
769
+ - Revenue Type: 6 values, total count: 227
770
+ - Video Game Title: 17 values, total count: 722
771
+ - Social Event Type: 5 values, total count: 217
772
+ - Purchase Channel: 4 values, total count: 316
773
+ - Harvest Method: 2 values, total count: 194
774
+ - Investment Asset Class: 5 values, total count: 329
775
+ - Family Member Count: 3 values, total count: 269
776
+ - Involvement Level: 2 values, total count: 191
777
+ - Payment Method: 11 values, total count: 331
778
+ - Fertilizer Type: 6 values, total count: 175
779
+ - Dwelling Type: 15 values, total count: 524
780
+ - Music Genre: 13 values, total count: 454
781
+ - Expense Type: 23 values, total count: 863
782
+ - Climate Zone Type: 4 values, total count: 168
783
+ - App Pricing Model: 1 values, total count: 90
784
+ - Vehicle Model: 28 values, total count: 678
785
+ - Ecosystem Type: 9 values, total count: 365
786
+ - Education Format: 2 values, total count: 174
787
+ - City: 21 values, total count: 517
788
+ - Medical Procedure: 12 values, total count: 425
789
+ - Produce Name: 11 values, total count: 344
790
+ - Landmark Name: 43 values, total count: 860
791
+ - Day Period: 4 values, total count: 264
792
+ - Building Style: 11 values, total count: 347
793
+ - Lighting Type: 3 values, total count: 134
794
+ - River Basin Name: 11 values, total count: 376
795
+ - Sales Metric: 1 values, total count: 80
796
+ - Recipient Category: 5 values, total count: 247
797
+ - Sport Position: 3 values, total count: 125
798
+ - Cover Crop Type: 15 values, total count: 403
799
+ - Generational Role: 6 values, total count: 300
800
+ - Turf Type: 2 values, total count: 156
801
+ - Occupational Role: 44 values, total count: 1,022
802
+ - Seafood Species: 31 values, total count: 896
803
+ - Metal Type: 5 values, total count: 267
804
+ - Marine Conservation Name: 17 values, total count: 451
805
+ - Protected Area Name: 37 values, total count: 807
806
+ - Travel Category: 6 values, total count: 308
807
+ - Career Level: 4 values, total count: 203
808
+ - Food Category: 7 values, total count: 301
809
+ - Tourism Type: 2 values, total count: 143
810
+ - Pandemic Period: 2 values, total count: 114
811
+ - Color: 14 values, total count: 507
812
+ - Galaxy Type: 4 values, total count: 225
813
+ - Air Quality Parameter: 5 values, total count: 186
814
+ - Social Media Intensity: 2 values, total count: 144
815
+ - Engagement: 2 values, total count: 144
816
+ - Cryptocurrency Name: 2 values, total count: 82
817
+ - Accessory Category: 9 values, total count: 372
818
+ - Launch Status: 2 values, total count: 142
819
+ - Jewelry Design Style: 8 values, total count: 371
820
+ - National Park Name: 10 values, total count: 287
821
+ - Legislative Status: 2 values, total count: 116
822
+ - Show Title: 8 values, total count: 286
823
+ - Art Movement: 4 values, total count: 119
824
+ - Health Coverage Type: 2 values, total count: 130
825
+ - Interest Category: 7 values, total count: 272
826
+ - Currency Code: 3 values, total count: 105
827
+ - Lake Name: 15 values, total count: 379
828
+ - Plant Establishment Method: 2 values, total count: 123
829
+ - Health Area: 4 values, total count: 175
830
+ - Mobile OS: 3 values, total count: 139
831
+ - Financial Incentive Type: 2 values, total count: 114
832
+ - Affectionate Gesture: 2 values, total count: 73
833
+ - Crypto Exchange: 6 values, total count: 255
834
+ - Higher Education Type: 2 values, total count: 119
835
+ - Medical Specialty: 22 values, total count: 592
836
+ - Eating Occasion: 4 values, total count: 206
837
+ - Music Subgenre: 6 values, total count: 202
838
+ - Storage Type: 4 values, total count: 118
839
+ - Battleground State: 2 values, total count: 76
840
+ - Zoning District: 6 values, total count: 207
841
+ - Data Collection Method: 6 values, total count: 130
842
+ - Season Phase: 2 values, total count: 93
843
+ - Phone Model: 16 values, total count: 418
844
+ - Environmental Impact: 7 values, total count: 224
845
+ - Coffee Style: 6 values, total count: 150
846
+ - Education Category: 17 values, total count: 330
847
+ - Game Name: 30 values, total count: 674
848
+ - Holiday Name: 8 values, total count: 251
849
+ - Waste Material: 4 values, total count: 164
850
+ - Museum Focus: 3 values, total count: 139
851
+ - Cargo Type: 3 values, total count: 124
852
+ - Game Genre: 8 values, total count: 228
853
+ - Acquisition Channel: 3 values, total count: 87
854
+ - Louisiana Parish: 8 values, total count: 242
855
+ - Conservation Status: 4 values, total count: 175
856
+ - Agricultural Practice: 16 values, total count: 370
857
+ - Funding Source: 4 values, total count: 142
858
+ - Supply Chain Stage: 11 values, total count: 271
859
+ - Application Name: 1 values, total count: 52
860
+ - Engineering Structure Type: 4 values, total count: 116
861
+ - Recipe Ingredient: 11 values, total count: 308
862
+ - Sofa Type: 3 values, total count: 110
863
+ - Disruption Reason: 4 values, total count: 84
864
+ - Financial Product Type: 6 values, total count: 226
865
+ - Festival Theme: 6 values, total count: 170
866
+ - Camera Lens Type: 2 values, total count: 98
867
+ - Sales Offer Type: 3 values, total count: 92
868
+ - Concession Product: 5 values, total count: 151
869
+ - Primate Species: 5 values, total count: 144
870
+ - Climate Type: 7 values, total count: 188
871
+ - Listening Method: 4 values, total count: 120
872
+ - Sports Facility Type: 7 values, total count: 236
873
+ - Performance Standing: 2 values, total count: 75
874
+ - Mission Complexity Level: 3 values, total count: 140
875
+ - Innovation Area: 4 values, total count: 129
876
+ - Parental Involvement Level: 2 values, total count: 76
877
+ - Sustainability Category: 5 values, total count: 96
878
+ - Travel Segment: 3 values, total count: 137
879
+ - Grain Category: 7 values, total count: 144
880
+ - Enforcement Degree: 3 values, total count: 138
881
+ - Fitness Goal: 5 values, total count: 134
882
+ - Light Pollution Severity: 2 values, total count: 86
883
+ - Social Determinant: 5 values, total count: 120
884
+ - Interaction Channel: 3 values, total count: 76
885
+ - Probability Basis: 2 values, total count: 58
886
+ - Medical Condition: 8 values, total count: 246
887
+ - Tourist Attraction: 8 values, total count: 190
888
+ - Sanction Measure: 4 values, total count: 106
889
+ - Furniture Category: 8 values, total count: 220
890
+ - Beverage Type: 18 values, total count: 398
891
+ - Education Level: 3 values, total count: 121
892
+ - Heritage Site Type: 9 values, total count: 179
893
+ - Earring Type: 5 values, total count: 179
894
+ - Animal Type: 6 values, total count: 172
895
+ - Achievement Award Category: 6 values, total count: 167
896
+ - Drug Name: 5 values, total count: 127
897
+ - Photography Genre: 4 values, total count: 126
898
+ - Developmental Skill Category: 2 values, total count: 83
899
+ - Alcoholic Beverage Type: 2 values, total count: 78
900
+ - Research Publication Type: 3 values, total count: 72
901
+ - Feedstock Type: 5 values, total count: 150
902
+ - Flyway Region: 4 values, total count: 138
903
+ - Usage Scenario: 6 values, total count: 137
904
+ - Adoption Level: 2 values, total count: 82
905
+ - Debris Material: 4 values, total count: 117
906
+ - Content Section: 4 values, total count: 110
907
+ - Design Style: 3 values, total count: 61
908
+ - Establishment Type: 6 values, total count: 152
909
+ - Membership Category: 4 values, total count: 136
910
+ - Satellite Function: 3 values, total count: 93
911
+ - Building Siding Material: 2 values, total count: 65
912
+ - Propulsion Type: 10 values, total count: 231
913
+ - Health Condition Category: 4 values, total count: 104
914
+ - Property Flooring Type: 2 values, total count: 78
915
+ - Connected Service: 4 values, total count: 75
916
+ - Fantasy Entity: 6 values, total count: 190
917
+ - Capture Method: 7 values, total count: 171
918
+ - Outreach Channel: 7 values, total count: 164
919
+ - Store Category: 6 values, total count: 127
920
+ - Marketing Channel: 6 values, total count: 123
921
+ - Data Domain: 7 values, total count: 146
922
+ - Art Subject: 3 values, total count: 104
923
+ - Cyber Attack Type: 3 values, total count: 75
924
+ - IP Infringement Type: 3 values, total count: 73
925
+ - Milestone Type: 2 values, total count: 54
926
+ - Medicine System: 2 values, total count: 48
927
+ - Major Sports Event: 8 values, total count: 197
928
+ - Citrus Variety: 5 values, total count: 151
929
+ - Insurance Type: 5 values, total count: 150
930
+ - Food Item: 4 values, total count: 129
931
+ - Award Ceremony: 7 values, total count: 146
932
+ - Tomato Variety: 6 values, total count: 126
933
+ - Mitigation Method: 9 values, total count: 152
934
+ - Insect Common Name: 3 values, total count: 99
935
+ - Driving Context: 3 values, total count: 91
936
+ - Funding Type: 4 values, total count: 84
937
+ - Price Level: 2 values, total count: 74
938
+ - Venue Seating Section: 1 values, total count: 37
939
+ - Cancer Type: 6 values, total count: 161
940
+ - Home Decor Category: 9 values, total count: 155
941
+ - Application Category: 6 values, total count: 148
942
+ - Music Distribution Format: 5 values, total count: 145
943
+ - Wearable Item Type: 5 values, total count: 127
944
+ - Court Surface: 3 values, total count: 96
945
+ - Play Type: 10 values, total count: 207
946
+ - Reef Site: 8 values, total count: 163
947
+ - Gemstone Color: 7 values, total count: 191
948
+ - Lighting Fixture Type: 6 values, total count: 148
949
+ - Livestock Category: 4 values, total count: 88
950
+ - Cultivation Environment: 2 values, total count: 70
951
+ - Sport Tier: 3 values, total count: 65
952
+ - Blockchain Platform: 1 values, total count: 35
953
+ - Wine Classification: 8 values, total count: 207
954
+ - Participation Level: 8 values, total count: 170
955
+ - Exercise Type: 7 values, total count: 162
956
+ - Content Streaming Platform: 8 values, total count: 158
957
+ - Email Provider: 5 values, total count: 142
958
+ - Sensor Category: 7 values, total count: 124
959
+ - Student Proficiency: 3 values, total count: 102
960
+ - Costume Character: 4 values, total count: 101
961
+ - Manure Management Method: 3 values, total count: 68
962
+ - Class Classification: 15 values, total count: 313
963
+ - Historical Period: 12 values, total count: 267
964
+ - Vineyard Designation: 7 values, total count: 134
965
+ - Care Level: 3 values, total count: 88
966
+ - Religious Symbol: 3 values, total count: 85
967
+ - Manufacturing Step: 5 values, total count: 85
968
+ - Pollution Source: 3 values, total count: 81
969
+ - Permit Sector: 3 values, total count: 66
970
+ - Cost Element: 2 values, total count: 56
971
+ - Fruit Name: 12 values, total count: 220
972
+ - Neighborhood: 11 values, total count: 214
973
+ - Amusement Ride Type: 7 values, total count: 142
974
+ - Dietary Food Group: 4 values, total count: 89
975
+ - Agricultural Region: 3 values, total count: 63
976
+ - Narrative Device: 2 values, total count: 46
977
+ - Gaming Publication: 7 values, total count: 173
978
+ - Economic Region: 8 values, total count: 132
979
+ - Retailer Name: 7 values, total count: 117
980
+ - Pesticide Classification: 4 values, total count: 116
981
+ - Dish Type: 7 values, total count: 114
982
+ - Impact Area: 5 values, total count: 114
983
+ - Painting Medium: 4 values, total count: 107
984
+ - Vehicle Maintenance Type: 5 values, total count: 104
985
+ - Seabird Species: 6 values, total count: 100
986
+ - Professional Skill: 6 values, total count: 91
987
+ - Campaign Type: 4 values, total count: 88
988
+ - Remediation Method: 3 values, total count: 69
989
+ - Insulation Material: 5 values, total count: 105
990
+ - Device Capability: 1 values, total count: 31
991
+ - Performance Indicator: 13 values, total count: 238
992
+ - Craft Supply: 11 values, total count: 224
993
+ - Land Management Practice: 6 values, total count: 149
994
+ - Vessel Classification: 8 values, total count: 141
995
+ - Shipping Passage: 4 values, total count: 112
996
+ - Forest Cover Type: 5 values, total count: 106
997
+ - Comic Issue: 5 values, total count: 97
998
+ - Geographic Region: 8 values, total count: 153
999
+ - GDP Component Type: 4 values, total count: 91
1000
+ - Shot Zone: 5 values, total count: 88
1001
+ - Irrigation Source Type: 3 values, total count: 75
1002
+ - Project Phase: 3 values, total count: 69
1003
+ - Sport Gender Division: 2 values, total count: 60
1004
+ - Water Supply Type: 2 values, total count: 60
1005
+ - Video Platform: 2 values, total count: 55
1006
+ - Index Basis: 2 values, total count: 54
1007
+ - Dress Silhouette: 2 values, total count: 42
1008
+ - Sacred Site Name: 1 values, total count: 30
1009
+ - Gift Category: 6 values, total count: 107
1010
+ - Gallery Sector: 2 values, total count: 53
1011
+ - App Usage Level: 3 values, total count: 85
1012
+ - Hurricane Wind Category: 4 values, total count: 82
1013
+ - Mitigation Project Type: 3 values, total count: 73
1014
+ - Agricultural Type: 6 values, total count: 119
1015
+ - Hub City: 2 values, total count: 58
1016
+ - Internet Access Status: 2 values, total count: 58
1017
+ - Tariff Active: 2 values, total count: 58
1018
+ - Patrol Method: 2 values, total count: 52
1019
+ - Reef Region: 3 values, total count: 49
1020
+ - Performance Terrain: 2 values, total count: 42
1021
+ - Pest Category: 11 values, total count: 235
1022
+ - Measure Type: 5 values, total count: 118
1023
+ - Workout Modality: 4 values, total count: 94
1024
+ - Impact Severity: 2 values, total count: 56
1025
+ - Sustainable Feature: 3 values, total count: 84
1026
+ - Green Space Accessibility: 3 values, total count: 84
1027
+ - Library Name: 2 values, total count: 50
1028
+ - Subscription Action Type: 2 values, total count: 38
1029
+ - Customer Segment Name: 16 values, total count: 254
1030
+ - Fatal Incident Type: 8 values, total count: 141
1031
+ - Tractor Configuration: 4 values, total count: 104
1032
+ - Service Category: 13 values, total count: 210
1033
+ - Accessibility Feature Type: 5 values, total count: 99
1034
+ - Consumer Profile: 4 values, total count: 96
1035
+ - Resource Classification: 3 values, total count: 78
1036
+ - Manufacturing Process: 4 values, total count: 74
1037
+ - Financial Concern Category: 3 values, total count: 69
1038
+ - Control Technique: 3 values, total count: 59
1039
+ - Weightlifting Exercise: 6 values, total count: 135
1040
+ - Software Application: 5 values, total count: 106
1041
+ - Support Category: 4 values, total count: 81
1042
+ - Therapeutic Area: 7 values, total count: 126
1043
+ - Orchard Region: 3 values, total count: 78
1044
+ - Access Level: 2 values, total count: 52
1045
+ - Result Status: 2 values, total count: 52
1046
+ - Subsidy Indicator: 2 values, total count: 52
1047
+ - Participation Status: 4 values, total count: 74
1048
+ - Japan Prefecture: 2 values, total count: 46
1049
+ - Worker Type: 2 values, total count: 37
1050
+ - Disaster Event: 10 values, total count: 197
1051
+ - Event Category: 5 values, total count: 101
1052
+ - Packaging Material: 7 values, total count: 125
1053
+ - Filtration Technology: 4 values, total count: 65
1054
+ - Election Type: 3 values, total count: 55
1055
+ - Wellness Program Type: 3 values, total count: 51
1056
+ - Marine Area Name: 11 values, total count: 193
1057
+ - Public Policy: 6 values, total count: 124
1058
+ - Market Trend: 9 values, total count: 139
1059
+ - Military Platform Type: 8 values, total count: 124
1060
+ - ADAS Module: 6 values, total count: 118
1061
+ - Rover Name: 5 values, total count: 104
1062
+ - Organization Sector: 4 values, total count: 75
1063
+ - Ballet Title: 4 values, total count: 75
1064
+ - Subscription Plan: 3 values, total count: 52
1065
+ - Footwear Style: 3 values, total count: 72
1066
+ - Environmental Strategy: 4 values, total count: 71
1067
+ - Space Mission: 3 values, total count: 64
1068
+ - National Leader: 2 values, total count: 48
1069
+ - Aircraft Model: 11 values, total count: 198
1070
+ - Project Category: 7 values, total count: 108
1071
+ - News Organization: 13 values, total count: 184
1072
+ - Zone Designation: 4 values, total count: 62
1073
+ - Pruning Style: 3 values, total count: 61
1074
+ - Clinical Service: 3 values, total count: 59
1075
+ - Hat Style: 3 values, total count: 58
1076
+ - Forage Type: 3 values, total count: 53
1077
+ - Ranching Approach: 2 values, total count: 33
1078
+ - Corridor Name: 8 values, total count: 110
1079
+ - Project Type: 7 values, total count: 121
1080
+ - Service Branch: 5 values, total count: 100
1081
+ - Source Collection: 5 values, total count: 86
1082
+ - App Permission: 5 values, total count: 86
1083
+ - Production Type: 5 values, total count: 82
1084
+ - Home Feature: 4 values, total count: 77
1085
+ - Booking Point: 4 values, total count: 76
1086
+ - Education Focus: 4 values, total count: 67
1087
+ - Loan Application Status: 3 values, total count: 54
1088
+ - Ecological Area Type: 3 values, total count: 53
1089
+ - Educational Attainment: 3 values, total count: 51
1090
+ - Pricing Basis: 2 values, total count: 44
1091
+ - Route Type: 2 values, total count: 37
1092
+ - System Component: 12 values, total count: 174
1093
+ - Conservation Organization: 9 values, total count: 146
1094
+ - Aid Sector: 5 values, total count: 105
1095
+ - Flavor Profile: 4 values, total count: 73
1096
+ - Essential Need: 3 values, total count: 63
1097
+ - Craft Material: 4 values, total count: 63
1098
+ - Fashion Style: 8 values, total count: 125
1099
+ - Character Kind: 3 values, total count: 45
1100
+ - Aircraft Category: 2 values, total count: 42
1101
+ - Flag State Name: 5 values, total count: 100
1102
+ - Roofing Type: 6 values, total count: 98
1103
+ - Technology Solution: 8 values, total count: 96
1104
+ - Fishery Location: 5 values, total count: 84
1105
+ - Amphibian Species Name: 5 values, total count: 81
1106
+ - Ballpark Name: 6 values, total count: 78
1107
+ - Tributary Name: 4 values, total count: 71
1108
+ - Instructional Method: 5 values, total count: 76
1109
+ - Mammal Species Name: 3 values, total count: 60
1110
+ - Insect Type: 3 values, total count: 60
1111
+ - Transit Station: 5 values, total count: 70
1112
+ - Theater Format: 2 values, total count: 40
1113
+ - Love Language Type: 2 values, total count: 40
1114
+ - Pollinator Classification: 2 values, total count: 40
1115
+ - Treatment Type: 2 values, total count: 31
1116
+ - Food Product Name: 10 values, total count: 134
1117
+ - Work Gear Type: 5 values, total count: 87
1118
+ - Space Agency: 4 values, total count: 76
1119
+ - Microphone Classification: 4 values, total count: 70
1120
+ - Management Technique: 4 values, total count: 63
1121
+ - Product Form: 4 values, total count: 60
1122
+ - Washing Machine Form Factor: 4 values, total count: 56
1123
+ - Subject Category: 3 values, total count: 50
1124
+ - Conflict Country: 3 values, total count: 47
1125
+ - Artwork Title: 2 values, total count: 32
1126
+ - Employment Model: 2 values, total count: 36
1127
+ - Computer Peripheral: 4 values, total count: 57
1128
+ - Farm Activity: 3 values, total count: 54
1129
+ - Agricultural Pest: 3 values, total count: 51
1130
+ - Federal Agency: 2 values, total count: 35
icon_generation/backup/extract_accepted_domains.py ADDED
@@ -0,0 +1,171 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """
3
+ 从 domain_value_pairs_enhanced.txt 中提取 accepted_domains.txt 中的 domains
4
+ 并按指定格式组织输出
5
+ """
6
+
7
+ from collections import defaultdict
8
+ from typing import Set, Dict, List
9
+
10
+
11
+ def parse_csv_line(line: str) -> tuple:
12
+ """解析 CSV 行"""
13
+ parts = []
14
+ current = []
15
+ in_quotes = False
16
+
17
+ for char in line:
18
+ if char == '"':
19
+ in_quotes = not in_quotes
20
+ elif char == ',' and not in_quotes:
21
+ parts.append(''.join(current).strip())
22
+ current = []
23
+ else:
24
+ current.append(char)
25
+ parts.append(''.join(current).strip())
26
+
27
+ if len(parts) >= 3:
28
+ domain = parts[0]
29
+ attribute = parts[1]
30
+ freq = int(parts[2])
31
+ return domain, attribute, freq
32
+ else:
33
+ return None, None, 0
34
+
35
+
36
+ def load_accepted_domains(file_path: str) -> Set[str]:
37
+ """加载可接受的 domains"""
38
+ domains = set()
39
+ with open(file_path, 'r', encoding='utf-8') as f:
40
+ for line in f:
41
+ line = line.strip()
42
+ if line:
43
+ domains.add(line)
44
+ return domains
45
+
46
+
47
+ def extract_accepted_domains(
48
+ input_file: str,
49
+ accepted_domains_file: str,
50
+ output_file: str = None
51
+ ):
52
+ """
53
+ 提取 accepted domains 的 domain-attribute pairs
54
+
55
+ Args:
56
+ input_file: 输入的 domain_value_pairs_enhanced.txt
57
+ accepted_domains_file: accepted_domains.txt
58
+ output_file: 输出文件路径(可选)
59
+ """
60
+ print("=" * 80)
61
+ print("提取 Accepted Domains 的 Domain-Attribute Pairs")
62
+ print("=" * 80)
63
+ print()
64
+
65
+ # 1. 加载可接受的 domains
66
+ print(f"📖 读取可接受的 domains: {accepted_domains_file}")
67
+ accepted_domains = load_accepted_domains(accepted_domains_file)
68
+ print(f"✅ 可接受的 domains: {len(accepted_domains)} 个")
69
+ print()
70
+
71
+ # 2. 读取并过滤 domain-attribute pairs
72
+ print(f"📖 读取 domain-attribute pairs: {input_file}")
73
+ domain_attributes = defaultdict(list)
74
+
75
+ with open(input_file, 'r', encoding='utf-8') as f:
76
+ for line in f:
77
+ line = line.strip()
78
+ if not line:
79
+ continue
80
+
81
+ domain, attribute, freq = parse_csv_line(line)
82
+ if domain and domain in accepted_domains:
83
+ domain_attributes[domain].append((attribute, freq))
84
+
85
+ print(f"✅ 找到 {len(domain_attributes)} 个匹配的 domains")
86
+ print()
87
+
88
+ # 3. 对每个 domain 的 attributes 按 frequency 排序
89
+ for domain in domain_attributes:
90
+ domain_attributes[domain].sort(key=lambda x: x[1], reverse=True)
91
+
92
+ # 4. 按 accepted_domains.txt 的顺序排序 domains
93
+ # 保持 accepted_domains.txt 中的顺序
94
+ accepted_list = []
95
+ with open(accepted_domains_file, 'r', encoding='utf-8') as f:
96
+ for line in f:
97
+ domain = line.strip()
98
+ if domain and domain in domain_attributes:
99
+ accepted_list.append(domain)
100
+
101
+ # 添加任何在数据中但不在列表中的 domains(以防万一)
102
+ for domain in domain_attributes:
103
+ if domain not in accepted_list:
104
+ accepted_list.append(domain)
105
+
106
+ # 5. 生成输出
107
+ print("=" * 80)
108
+ print("📊 生成的 Domain-Attribute 列表:")
109
+ print("=" * 80)
110
+ print()
111
+
112
+ output_lines = []
113
+
114
+ for domain in accepted_list:
115
+ if domain not in domain_attributes:
116
+ continue
117
+
118
+ attributes = domain_attributes[domain]
119
+ attr_names = [attr for attr, _ in attributes]
120
+ attr_str = ", ".join(attr_names)
121
+
122
+ # 添加到输出
123
+ output_lines.append(domain)
124
+ output_lines.append(attr_str)
125
+ output_lines.append("") # 空行
126
+
127
+ # 打印到控制台
128
+ print(domain)
129
+ print(attr_str)
130
+ print()
131
+
132
+ # 6. 保存到文件(如果指定)
133
+ if output_file:
134
+ print("=" * 80)
135
+ print(f"💾 保存到文件: {output_file}")
136
+
137
+ with open(output_file, 'w', encoding='utf-8') as f:
138
+ for line in output_lines:
139
+ f.write(line + "\n")
140
+
141
+ print(f"✅ 已保存")
142
+
143
+ # 7. 统计信息
144
+ print()
145
+ print("=" * 80)
146
+ print("📈 统计信息:")
147
+ print("=" * 80)
148
+ print(f"匹配的 Domains 数: {len(domain_attributes)}")
149
+ print(f"总 Attributes 数: {sum(len(attrs) for attrs in domain_attributes.values())}")
150
+
151
+ # 显示每个 domain 的 attributes 数量
152
+ print()
153
+ print("各 Domain 的 Attributes 数量:")
154
+ for domain in accepted_list:
155
+ if domain in domain_attributes:
156
+ count = len(domain_attributes[domain])
157
+ print(f" {domain}: {count} attributes")
158
+
159
+ print()
160
+ print("=" * 80)
161
+ print("✨ 提取完成!")
162
+ print("=" * 80)
163
+
164
+
165
+ if __name__ == '__main__':
166
+ input_file = 'domain_value_pairs_enhanced.txt'
167
+ accepted_domains_file = 'accepted_domains.txt'
168
+ output_file = 'accepted_domains_attributes.txt'
169
+
170
+ extract_accepted_domains(input_file, accepted_domains_file, output_file)
171
+
icon_generation/backup/filter_accepted_domains.py ADDED
@@ -0,0 +1,445 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """
3
+ 过滤和精简 accepted_domains_attributes.txt
4
+ 1. 删除过于小众的 domains
5
+ 2. 对于 attributes 不容易区分的 domain,只保留 5-10 个最重要的
6
+ 3. 大多数正常的 domain 完全保留
7
+ """
8
+
9
+ import json
10
+ import os
11
+ import requests
12
+ from typing import Dict, Optional, List, Tuple
13
+ from concurrent.futures import ThreadPoolExecutor, as_completed
14
+ import threading
15
+ import time
16
+
17
+
18
+ class DomainFilter:
19
+ """使用 LLM 过滤和精简 domains"""
20
+
21
+ def __init__(self, api_key=None, base_url=None, model=None):
22
+ """初始化 LLM analyzer"""
23
+ self.api_key = api_key or os.getenv("OPENAI_API_KEY") or os.getenv("AIHUBMIX_API_KEY", "")
24
+ self.base_url = base_url or os.getenv("OPENAI_BASE_URL", "https://aihubmix.com/v1")
25
+ self.model = model or os.getenv("OPENAI_MODEL", "gpt-5.2")
26
+ self.lock = threading.Lock()
27
+ self.completed_count = 0
28
+ self.total_domains = 0
29
+
30
+ def filter_domains_batch(self, domains_data: List[Tuple[str, List[str]]], idx: int, total: int, start_time: float) -> Dict:
31
+ """
32
+ 批量处理 domains(每次 10 个)
33
+
34
+ Args:
35
+ domains_data: [(domain, [attributes]), ...] 最多 10 个
36
+ idx: 批次索引
37
+ total: 总批次数
38
+ start_time: 开始时间
39
+
40
+ Returns:
41
+ {
42
+ 'keep': [(domain, [attributes]), ...],
43
+ 'remove': [domain, ...],
44
+ 'reasoning': 'explanation'
45
+ }
46
+ """
47
+ prompt = self._build_filter_prompt(domains_data)
48
+
49
+ with self.lock:
50
+ progress = (idx / total) * 100
51
+ elapsed = time.time() - start_time
52
+ avg_time = elapsed / idx if idx > 0 else 0
53
+ remaining = avg_time * (total - idx)
54
+
55
+ print(f"\n{'=' * 80}")
56
+ print(f"🔄 处理批次 {idx}/{total} ({progress:.1f}%)")
57
+ print(f"⏱️ 已用时间: {elapsed:.1f}秒 | 预计剩余: {remaining:.1f}秒")
58
+ print(f" 当前批次: {len(domains_data)} 个 domains")
59
+ print(f" 🤖 调用 LLM 进行过滤和精简...")
60
+
61
+ try:
62
+ response = self._query_llm(prompt)
63
+
64
+ if response:
65
+ # 清理可能的 markdown 代码块
66
+ cleaned_response = response.strip()
67
+ if cleaned_response.startswith('```'):
68
+ lines = cleaned_response.split('\n')
69
+ cleaned_response = '\n'.join(lines[1:-1] if lines[-1].strip() == '```' else lines[1:])
70
+ cleaned_response = cleaned_response.replace('```json', '').replace('```', '').strip()
71
+
72
+ result = json.loads(cleaned_response)
73
+
74
+ # 验证返回格式
75
+ if 'domains' in result:
76
+ keep = []
77
+ remove = []
78
+
79
+ for item in result['domains']:
80
+ domain = item['domain']
81
+ action = item.get('action', 'keep')
82
+
83
+ if action == 'remove':
84
+ remove.append(domain)
85
+ elif action == 'keep_all':
86
+ # 找到原始数据
87
+ for orig_domain, orig_attrs in domains_data:
88
+ if orig_domain == domain:
89
+ keep.append((domain, orig_attrs))
90
+ break
91
+ elif action == 'keep_reduced':
92
+ # 使用精简后的 attributes
93
+ reduced_attrs = item.get('reduced_attributes', [])
94
+ if reduced_attrs:
95
+ keep.append((domain, reduced_attrs))
96
+ else:
97
+ # 如果没有提供,保留原始
98
+ for orig_domain, orig_attrs in domains_data:
99
+ if orig_domain == domain:
100
+ keep.append((domain, orig_attrs))
101
+ break
102
+ else:
103
+ # 默认保留
104
+ for orig_domain, orig_attrs in domains_data:
105
+ if orig_domain == domain:
106
+ keep.append((domain, orig_attrs))
107
+ break
108
+
109
+ with self.lock:
110
+ self.completed_count += 1
111
+ print(f" ✅ 处理完成!")
112
+ print(f" 保留: {len(keep)} 个")
113
+ print(f" 删除: {len(remove)} 个")
114
+ if remove:
115
+ print(f" 删除的 domains: {', '.join(remove)}")
116
+
117
+ return {
118
+ 'keep': keep,
119
+ 'remove': remove,
120
+ 'reasoning': result.get('reasoning', '')
121
+ }
122
+ else:
123
+ with self.lock:
124
+ self.completed_count += 1
125
+ print(f" ⚠️ LLM 响应缺少必需字段,保留所有")
126
+ # 默认全部保留
127
+ return {
128
+ 'keep': [(domain, attrs) for domain, attrs in domains_data],
129
+ 'remove': [],
130
+ 'reasoning': 'LLM response format error, kept all'
131
+ }
132
+
133
+ else:
134
+ with self.lock:
135
+ self.completed_count += 1
136
+ print(f" ⚠️ LLM API 调用失败,保留所有")
137
+ return {
138
+ 'keep': [(domain, attrs) for domain, attrs in domains_data],
139
+ 'remove': [],
140
+ 'reasoning': 'LLM API failed, kept all'
141
+ }
142
+
143
+ except json.JSONDecodeError as e:
144
+ with self.lock:
145
+ self.completed_count += 1
146
+ print(f" ⚠️ LLM 响应不是有效的 JSON: {e}")
147
+ return {
148
+ 'keep': [(domain, attrs) for domain, attrs in domains_data],
149
+ 'remove': [],
150
+ 'reasoning': 'JSON decode error, kept all'
151
+ }
152
+ except Exception as e:
153
+ with self.lock:
154
+ self.completed_count += 1
155
+ print(f" ⚠️ 处理错误: {e}")
156
+ return {
157
+ 'keep': [(domain, attrs) for domain, attrs in domains_data],
158
+ 'remove': [],
159
+ 'reasoning': f'Error: {str(e)}, kept all'
160
+ }
161
+
162
+ def _build_filter_prompt(self, domains_data: List[Tuple[str, List[str]]]) -> str:
163
+ """构建过滤的 prompt"""
164
+
165
+ domains_str = ""
166
+ for domain, attributes in domains_data:
167
+ attrs_str = ", ".join(attributes[:20]) # 最多显示前20个
168
+ if len(attributes) > 20:
169
+ attrs_str += f", ... (共 {len(attributes)} 个)"
170
+ domains_str += f"\nDomain: {domain}\nAttributes: {attrs_str}\n"
171
+
172
+ prompt = f"""You are a data curation expert. Given a list of domains and their attributes, your task is to:
173
+
174
+ 1. **REMOVE** domains that are too niche, specialized, or not commonly used (e.g., "Lemur Taxon Name (Genus and Species)", highly technical scientific classifications, extremely specific subcategories)
175
+ 2. **KEEP ALL** domains that are mainstream, widely recognized, and useful for general purposes (e.g., "Industry Sector", "Land Use Type", "Product Category", "Transportation Mode")
176
+ 3. **REDUCE** attributes for domains where attributes are hard to distinguish or too granular (e.g., "Donor Classification" with "Individual Donor", "One-time Donor", "Recurring Donor", "Monthly Donor" - these are too similar). For such domains, keep only 5-10 most important and distinct attributes.
177
+
178
+ Guidelines:
179
+ - **REMOVE** if the domain is:
180
+ - Too specific or scientific (e.g., species names, technical taxonomies)
181
+ - Too niche or rarely used in general contexts
182
+ - Overly granular subcategories
183
+
184
+ - **KEEP ALL** if the domain is:
185
+ - Commonly used in business, data analysis, or general applications
186
+ - Well-known categories (e.g., industries, product types, locations)
187
+ - Useful for visualization or categorization
188
+
189
+ - **REDUCE** attributes if:
190
+ - Attributes are very similar or hard to distinguish
191
+ - Too many granular variations (e.g., "Monthly Donor" vs "Recurring Donor")
192
+ - Keep 5-10 most important and distinct ones
193
+
194
+ Domains to evaluate:
195
+ {domains_str}
196
+
197
+ Return your response in the following JSON format ONLY (no markdown, no extra text):
198
+ {{
199
+ "domains": [
200
+ {{
201
+ "domain": "Domain Name",
202
+ "action": "remove",
203
+ "reduced_attributes": []
204
+ }},
205
+ {{
206
+ "domain": "Domain Name",
207
+ "action": "keep_all",
208
+ "reduced_attributes": []
209
+ }},
210
+ {{
211
+ "domain": "Domain Name",
212
+ "action": "keep_reduced",
213
+ "reduced_attributes": ["attr1", "attr2", "attr3"]
214
+ }}
215
+ ],
216
+ "reasoning": "Brief explanation of decisions"
217
+ }}
218
+
219
+ Examples:
220
+ - "Lemur Taxon Name (Genus and Species)" → action: "remove" (too niche)
221
+ - "Industry Sector" → action: "keep_all" (mainstream, useful)
222
+ - "Donor Classification" with many similar attributes → action: "keep_reduced", reduced_attributes: ["Individual", "Corporate", "Foundation", "Government", "Anonymous"]
223
+ """
224
+
225
+ return prompt
226
+
227
+ def _query_llm(self, prompt: str) -> Optional[str]:
228
+ """查询 LLM API"""
229
+ headers = {
230
+ 'Authorization': f'Bearer {self.api_key}',
231
+ 'Content-Type': 'application/json'
232
+ }
233
+
234
+ data = {
235
+ 'model': self.model,
236
+ 'messages': [
237
+ {
238
+ 'role': 'system',
239
+ 'content': 'You are a data curation expert specialized in filtering and organizing domain-attribute pairs. Always return valid JSON format only, without any markdown formatting or extra text.'
240
+ },
241
+ {
242
+ 'role': 'user',
243
+ 'content': prompt
244
+ }
245
+ ],
246
+ 'temperature': 0.3
247
+ }
248
+
249
+ try:
250
+ response = requests.post(
251
+ f'{self.base_url}/chat/completions',
252
+ headers=headers,
253
+ json=data,
254
+ timeout=60
255
+ )
256
+ response.raise_for_status()
257
+
258
+ result = response.json()
259
+ return result['choices'][0]['message']['content'].strip()
260
+
261
+ except requests.exceptions.Timeout:
262
+ with self.lock:
263
+ print(" ❌ LLM API 超时")
264
+ return None
265
+ except requests.exceptions.HTTPError as e:
266
+ with self.lock:
267
+ print(f" ❌ LLM API HTTP 错误: {e}")
268
+ return None
269
+ except requests.exceptions.RequestException as e:
270
+ with self.lock:
271
+ print(f" ❌ LLM API 请求错误: {e}")
272
+ return None
273
+ except KeyError as e:
274
+ with self.lock:
275
+ print(f" ❌ LLM API 响应格式错误: {e}")
276
+ return None
277
+
278
+
279
+ def load_domains_attributes(file_path: str) -> List[Tuple[str, List[str]]]:
280
+ """加载 domains 和 attributes"""
281
+ domains_data = []
282
+ current_domain = None
283
+ current_attrs = []
284
+
285
+ with open(file_path, 'r', encoding='utf-8') as f:
286
+ for line in f:
287
+ line = line.strip()
288
+ if not line:
289
+ # 空行表示一个 domain 结束
290
+ if current_domain:
291
+ domains_data.append((current_domain, current_attrs))
292
+ current_domain = None
293
+ current_attrs = []
294
+ continue
295
+
296
+ # 检查是否是新的 domain(没有逗号,且不是 attributes 行)
297
+ if ',' not in line and not line.startswith(' ') and len(line) > 0:
298
+ # 保存之前的 domain
299
+ if current_domain:
300
+ domains_data.append((current_domain, current_attrs))
301
+ # 开始新的 domain
302
+ current_domain = line
303
+ current_attrs = []
304
+ else:
305
+ # 这是 attributes 行
306
+ if current_domain:
307
+ attrs = [attr.strip() for attr in line.split(',')]
308
+ current_attrs.extend(attrs)
309
+
310
+ # 保存最后一个 domain
311
+ if current_domain:
312
+ domains_data.append((current_domain, current_attrs))
313
+
314
+ return domains_data
315
+
316
+
317
+ def main():
318
+ print("=" * 80)
319
+ print("Domain-Attribute 过滤和精简工具")
320
+ print("=" * 80)
321
+ print()
322
+
323
+ input_file = 'accepted_domains_attributes.txt'
324
+ output_file = 'accepted_domains_attributes_filtered.txt'
325
+
326
+ # 1. 加载数据
327
+ print(f"📖 读取文件: {input_file}")
328
+ domains_data = load_domains_attributes(input_file)
329
+ print(f"✅ 成功读取 {len(domains_data)} 个 domains")
330
+ print()
331
+
332
+ # 2. 初始化过滤器
333
+ filter_obj = DomainFilter()
334
+
335
+ # 3. 分批处理(每批 10 个)
336
+ batch_size = 10
337
+ batches = []
338
+ for i in range(0, len(domains_data), batch_size):
339
+ batch = domains_data[i:i+batch_size]
340
+ batches.append(batch)
341
+
342
+ total_batches = len(batches)
343
+ filter_obj.total_domains = total_batches
344
+
345
+ print("=" * 80)
346
+ print(f"🚀 开始处理 (10线程并发)")
347
+ print(f" 总 domains: {len(domains_data)}")
348
+ print(f" 批次数: {total_batches}")
349
+ print(f" 每批: {batch_size} 个 domains")
350
+ print("=" * 80)
351
+ print()
352
+
353
+ start_time = time.time()
354
+
355
+ # 4. 使用线程池并发处理
356
+ filtered_results = []
357
+ num_threads = 10
358
+
359
+ with ThreadPoolExecutor(max_workers=num_threads) as executor:
360
+ # 提交所有任务
361
+ future_to_batch = {
362
+ executor.submit(
363
+ filter_obj.filter_domains_batch,
364
+ batch,
365
+ idx + 1,
366
+ total_batches,
367
+ start_time
368
+ ): (idx, batch)
369
+ for idx, batch in enumerate(batches)
370
+ }
371
+
372
+ # 收集结果
373
+ for future in as_completed(future_to_batch):
374
+ idx, batch = future_to_batch[future]
375
+ try:
376
+ result = future.result()
377
+ filtered_results.append(result)
378
+ except Exception as e:
379
+ with filter_obj.lock:
380
+ print(f"❌ 处理批次 {idx + 1} 时出错: {e}")
381
+ # 默认保留
382
+ filtered_results.append({
383
+ 'keep': [(domain, attrs) for domain, attrs in batch],
384
+ 'remove': [],
385
+ 'reasoning': f'Error: {str(e)}'
386
+ })
387
+
388
+ # 5. 合并结果
389
+ print()
390
+ print("=" * 80)
391
+ print("📊 合并结果...")
392
+ print("=" * 80)
393
+
394
+ final_domains = []
395
+ removed_domains = []
396
+
397
+ for result in filtered_results:
398
+ final_domains.extend(result['keep'])
399
+ removed_domains.extend(result['remove'])
400
+
401
+ # 6. 保存结果
402
+ print()
403
+ print("=" * 80)
404
+ print("💾 保存结果...")
405
+ print("=" * 80)
406
+
407
+ with open(output_file, 'w', encoding='utf-8') as f:
408
+ for domain, attributes in final_domains:
409
+ f.write(f"{domain}\n")
410
+ f.write(f"{', '.join(attributes)}\n")
411
+ f.write("\n")
412
+
413
+ elapsed_time = time.time() - start_time
414
+
415
+ # 7. 统计信息
416
+ total_attributes = sum(len(attrs) for _, attrs in final_domains)
417
+
418
+ print()
419
+ print("=" * 80)
420
+ print("📊 统计信息:")
421
+ print("=" * 80)
422
+ print(f"原始 domains 数: {len(domains_data)}")
423
+ print(f"保留 domains 数: {len(final_domains)}")
424
+ print(f"删除 domains 数: {len(removed_domains)}")
425
+ print(f"总 attributes 数: {total_attributes}")
426
+ print(f"总耗时: {elapsed_time:.1f} 秒")
427
+ print()
428
+
429
+ if removed_domains:
430
+ print("删除的 domains:")
431
+ for domain in removed_domains[:20]: # 最多显示20个
432
+ print(f" - {domain}")
433
+ if len(removed_domains) > 20:
434
+ print(f" ... 还有 {len(removed_domains) - 20} 个")
435
+ print()
436
+
437
+ print(f"💾 结果已保存到: {output_file}")
438
+ print()
439
+ print("=" * 80)
440
+ print("✨ 处理完成!")
441
+ print("=" * 80)
442
+
443
+
444
+ if __name__ == '__main__':
445
+ main()
icon_generation/backup/filter_accepted_log.txt ADDED
@@ -0,0 +1,383 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ================================================================================
2
+ Domain-Attribute 过滤和精简工具
3
+ ================================================================================
4
+
5
+ 📖 读取文件: accepted_domains_attributes.txt
6
+ ✅ 成功读取 338 个 domains
7
+
8
+ ================================================================================
9
+ 🚀 开始处理 (10线程并发)
10
+ 总 domains: 338
11
+ 批次数: 34
12
+ 每批: 10 个 domains
13
+ ================================================================================
14
+
15
+
16
+ ================================================================================
17
+ 🔄 处理批次 1/34 (2.9%)
18
+ ⏱️ 已用时间: 0.0秒 | 预计剩余: 0.0秒
19
+ 当前批次: 10 个 domains
20
+ 🤖 调用 LLM 进行过滤和精简...
21
+
22
+ ================================================================================
23
+ 🔄 处理批次 2/34 (5.9%)
24
+ ⏱️ 已用时间: 0.0秒 | 预计剩余: 0.0秒
25
+ 当前批次: 10 个 domains
26
+ 🤖 调用 LLM 进行过滤和精简...
27
+
28
+ ================================================================================
29
+ 🔄 处理批次 3/34 (8.8%)
30
+ ⏱️ 已用时间: 0.0秒 | 预计剩余: 0.0秒
31
+ 当前批次: 10 个 domains
32
+ 🤖 调用 LLM 进行过滤和精简...
33
+
34
+ ================================================================================
35
+ 🔄 处理批次 4/34 (11.8%)
36
+ ⏱️ 已用时间: 0.0秒 | 预计剩余: 0.0秒
37
+ 当前批次: 10 个 domains
38
+ 🤖 调用 LLM 进行过滤和精简...
39
+
40
+ ================================================================================
41
+ 🔄 处理批次 5/34 (14.7%)
42
+ ⏱️ 已用时间: 0.0秒 | 预计剩余: 0.0秒
43
+ 当前批次: 10 个 domains
44
+ 🤖 调用 LLM 进行过滤和精简...
45
+
46
+ ================================================================================
47
+ 🔄 处理批次 6/34 (17.6%)
48
+ ⏱️ 已用时间: 0.0秒 | 预计剩余: 0.0秒
49
+ 当前批次: 10 个 domains
50
+ 🤖 调用 LLM 进行过滤和精简...
51
+
52
+ ================================================================================
53
+ 🔄 处理批次 7/34 (20.6%)
54
+ ⏱️ 已用时间: 0.0秒 | 预计剩余: 0.1秒
55
+ 当前批次: 10 个 domains
56
+ 🤖 调用 LLM 进行过滤和精简...
57
+
58
+ ================================================================================
59
+ 🔄 处理批次 8/34 (23.5%)
60
+ ⏱️ 已用时间: 0.0秒 | 预计剩余: 0.1秒
61
+ 当前批次: 10 个 domains
62
+ 🤖 调用 LLM 进行过滤和精简...
63
+
64
+ ================================================================================
65
+ 🔄 处理批次 9/34 (26.5%)
66
+ ⏱️ 已用时间: 0.0秒 | 预计剩余: 0.1秒
67
+ 当前批次: 10 个 domains
68
+ 🤖 调用 LLM 进行过滤和精简...
69
+
70
+ ================================================================================
71
+ 🔄 处理批次 10/34 (29.4%)
72
+ ⏱️ 已用时间: 0.0秒 | 预计剩余: 0.0秒
73
+ 当前批次: 10 个 domains
74
+ 🤖 调用 LLM 进行过滤和精简...
75
+ ✅ 处理完成!
76
+ 保留: 10 个
77
+ 删除: 0 个
78
+
79
+ ================================================================================
80
+ 🔄 处理批次 11/34 (32.4%)
81
+ ⏱️ 已用时间: 10.9秒 | 预计剩余: 22.8秒
82
+ 当前批次: 10 个 domains
83
+ 🤖 调用 LLM 进行过滤和精简...
84
+ ✅ 处理完成!
85
+ 保留: 10 个
86
+ 删除: 0 个
87
+
88
+ ================================================================================
89
+ 🔄 处理批次 12/34 (35.3%)
90
+ ⏱️ 已用时间: 11.2秒 | 预计剩余: 20.6秒
91
+ 当前批次: 10 个 domains
92
+ 🤖 调用 LLM 进行过滤和精简...
93
+ ✅ 处理完成!
94
+ 保留: 9 个
95
+ 删除: 1 个
96
+ 删除的 domains: Landmark Name
97
+
98
+ ================================================================================
99
+ 🔄 处理批次 13/34 (38.2%)
100
+ ⏱️ 已用时间: 11.4秒 | 预计剩余: 18.4秒
101
+ 当前批次: 10 个 domains
102
+ 🤖 调用 LLM 进行过滤和精简...
103
+ ✅ 处理完成!
104
+ 保留: 10 个
105
+ 删除: 0 个
106
+
107
+ ================================================================================
108
+ 🔄 处理批次 14/34 (41.2%)
109
+ ⏱️ 已用时间: 11.7秒 | 预计剩余: 16.7秒
110
+ 当前批次: 10 个 domains
111
+ 🤖 调用 LLM 进行过滤和精简...
112
+ ✅ 处理完成!
113
+ 保留: 10 个
114
+ 删除: 0 个
115
+
116
+ ================================================================================
117
+ 🔄 处理批次 15/34 (44.1%)
118
+ ⏱️ 已用时间: 12.2秒 | 预计剩余: 15.4秒
119
+ 当前批次: 10 个 domains
120
+ 🤖 调用 LLM 进行过滤和精简...
121
+ ✅ 处理完成!
122
+ 保留: 10 个
123
+ 删除: 0 个
124
+
125
+ ================================================================================
126
+ 🔄 处理批次 16/34 (47.1%)
127
+ ⏱️ 已用时间: 13.9秒 | 预计剩余: 15.6秒
128
+ 当前批次: 10 个 domains
129
+ 🤖 调用 LLM 进行过滤和精简...
130
+ ✅ 处理完成!
131
+ 保留: 10 个
132
+ 删除: 0 个
133
+
134
+ ================================================================================
135
+ 🔄 处理批次 17/34 (50.0%)
136
+ ⏱️ 已用时间: 14.1秒 | 预计剩余: 14.1秒
137
+ 当前批次: 10 个 domains
138
+ 🤖 调用 LLM 进行过滤和精简...
139
+ ✅ 处理完成!
140
+ 保留: 10 个
141
+ 删除: 0 个
142
+
143
+ ================================================================================
144
+ 🔄 处理批次 18/34 (52.9%)
145
+ ⏱️ 已用时间: 14.6秒 | 预计剩余: 13.0秒
146
+ 当前批次: 10 个 domains
147
+ 🤖 调用 LLM 进行过滤和精简...
148
+ ✅ 处理完成!
149
+ 保留: 10 个
150
+ 删除: 0 个
151
+
152
+ ================================================================================
153
+ 🔄 处理批次 19/34 (55.9%)
154
+ ⏱️ 已用时间: 18.8秒 | 预计剩余: 14.9秒
155
+ 当前批次: 10 个 domains
156
+ 🤖 调用 LLM 进行过滤和精简...
157
+ ✅ 处理完成!
158
+ 保留: 10 个
159
+ 删除: 0 个
160
+
161
+ ================================================================================
162
+ 🔄 处理批次 20/34 (58.8%)
163
+ ⏱️ 已用时间: 19.0秒 | 预计剩余: 13.3秒
164
+ 当前批次: 10 个 domains
165
+ 🤖 调用 LLM 进行过滤和精简...
166
+ ✅ 处理完成!
167
+ 保留: 9 个
168
+ 删除: 1 个
169
+ 删除的 domains: Citrus Variety
170
+
171
+ ================================================================================
172
+ 🔄 处理批次 21/34 (61.8%)
173
+ ⏱️ 已用时间: 23.1秒 | 预计剩余: 14.3秒
174
+ 当前批次: 10 个 domains
175
+ 🤖 调用 LLM 进行过滤和精简...
176
+ ✅ 处理完成!
177
+ 保留: 9 个
178
+ 删除: 1 个
179
+ 删除的 domains: Tourist Attraction
180
+
181
+ ================================================================================
182
+ 🔄 处理批次 22/34 (64.7%)
183
+ ⏱️ 已用时间: 25.6秒 | 预计剩余: 14.0秒
184
+ 当前批次: 10 个 domains
185
+ 🤖 调用 LLM 进行过滤和精简...
186
+ ✅ 处理完成!
187
+ 保留: 10 个
188
+ 删除: 0 个
189
+
190
+ ================================================================================
191
+ 🔄 处理批次 23/34 (67.6%)
192
+ ⏱️ 已用时间: 26.5秒 | 预计剩余: 12.7秒
193
+ 当前批次: 10 个 domains
194
+ 🤖 调用 LLM 进行过滤和精简...
195
+ ✅ 处理完成!
196
+ 保留: 10 个
197
+ 删除: 0 个
198
+
199
+ ================================================================================
200
+ 🔄 处理批次 24/34 (70.6%)
201
+ ⏱️ 已用时间: 29.2秒 | 预计剩余: 12.2秒
202
+ 当前批次: 10 个 domains
203
+ 🤖 调用 LLM 进行过滤和精简...
204
+ ✅ 处理完成!
205
+ 保留: 9 个
206
+ 删除: 1 个
207
+ 删除的 domains: Primate Species
208
+
209
+ ================================================================================
210
+ 🔄 处理批次 25/34 (73.5%)
211
+ ⏱️ 已用时间: 29.4秒 | 预计剩余: 10.6秒
212
+ 当前批次: 10 个 domains
213
+ 🤖 调用 LLM 进行过滤和精简...
214
+ ✅ 处理完成!
215
+ 保留: 8 个
216
+ 删除: 2 个
217
+ 删除的 domains: Satellite Mission Type, Fantasy Creature or Race
218
+
219
+ ================================================================================
220
+ 🔄 处理批次 26/34 (76.5%)
221
+ ⏱️ 已用时间: 30.0秒 | 预计剩余: 9.2秒
222
+ 当前批次: 10 个 domains
223
+ 🤖 调用 LLM 进行过滤和精简...
224
+ ✅ 处理完成!
225
+ 保留: 10 个
226
+ 删除: 0 个
227
+
228
+ ================================================================================
229
+ 🔄 处理批次 27/34 (79.4%)
230
+ ⏱️ 已用时间: 30.0秒 | 预计剩余: 7.8秒
231
+ 当前批次: 10 个 domains
232
+ 🤖 调用 LLM 进行过滤和精简...
233
+ ✅ 处理完成!
234
+ 保留: 9 个
235
+ 删除: 1 个
236
+ 删除的 domains: Notable Sacred Sites
237
+
238
+ ================================================================================
239
+ 🔄 处理批次 28/34 (82.4%)
240
+ ⏱️ 已用时间: 30.8秒 | 预计剩余: 6.6秒
241
+ 当前批次: 10 个 domains
242
+ 🤖 调用 LLM 进行过滤和精简...
243
+ ✅ 处理完成!
244
+ 保留: 10 个
245
+ 删除: 0 个
246
+
247
+ ================================================================================
248
+ 🔄 处理批次 29/34 (85.3%)
249
+ ⏱️ 已用时间: 33.5秒 | 预计剩余: 5.8秒
250
+ 当前批次: 10 个 domains
251
+ 🤖 调用 LLM 进行过滤和精简...
252
+ ✅ 处理完成!
253
+ 保留: 10 个
254
+ 删除: 0 个
255
+
256
+ ================================================================================
257
+ 🔄 处理批次 30/34 (88.2%)
258
+ ⏱️ 已用时间: 34.2秒 | 预计剩余: 4.6秒
259
+ 当前批次: 10 个 domains
260
+ 🤖 调用 LLM 进行过滤和精简...
261
+ ✅ 处理完成!
262
+ 保留: 8 个
263
+ 删除: 2 个
264
+ 删除的 domains: Planetary Rover Name, Commercial Aircraft Model
265
+
266
+ ================================================================================
267
+ 🔄 处理批次 31/34 (91.2%)
268
+ ⏱️ 已用时间: 37.1秒 | 预计剩余: 3.6秒
269
+ 当前批次: 10 个 domains
270
+ 🤖 调用 LLM 进行过滤和精简...
271
+ ✅ 处理完成!
272
+ 保留: 7 个
273
+ 删除: 3 个
274
+ 删除的 domains: Common Amphibian Species (common names), Mammal Species Name, Artwork Title
275
+
276
+ ================================================================================
277
+ 🔄 处理批次 32/34 (94.1%)
278
+ ⏱️ 已用时间: 38.7秒 | 预计剩余: 2.4秒
279
+ 当前批次: 10 个 domains
280
+ 🤖 调用 LLM 进行过滤和精简...
281
+ ✅ 处理完成!
282
+ 保留: 8 个
283
+ 删除: 2 个
284
+ 删除的 domains: Passenger Persona (Travel Behavior), Solar System Planet
285
+
286
+ ================================================================================
287
+ 🔄 处理批次 33/34 (97.1%)
288
+ ⏱️ 已用时间: 38.7秒 | 预计剩余: 1.2秒
289
+ 当前批次: 10 个 domains
290
+ 🤖 调用 LLM 进行过滤和精简...
291
+ ✅ 处理完成!
292
+ 保留: 10 个
293
+ 删除: 0 个
294
+
295
+ ================================================================================
296
+ 🔄 处理批次 34/34 (100.0%)
297
+ ⏱️ 已用时间: 39.2秒 | 预计剩余: 0.0秒
298
+ 当前批次: 8 个 domains
299
+ 🤖 调用 LLM 进行过滤和精简...
300
+ ✅ 处理完成!
301
+ 保留: 6 个
302
+ 删除: 4 个
303
+ 删除的 domains: Jewelry-making Technique, Cultural Attire, Chinese New Year Foods, Maize (Corn) Cultivar/Variety
304
+ ✅ 处理完成!
305
+ 保留: 9 个
306
+ 删除: 1 个
307
+ 删除的 domains: Ranching Practices and Systems
308
+ ✅ 处理完成!
309
+ 保留: 8 个
310
+ 删除: 2 个
311
+ 删除的 domains: Astronomical Observatory Name, Historic Site Name
312
+ ✅ 处理完成!
313
+ 保留: 7 个
314
+ 删除: 3 个
315
+ 删除的 domains: Role-Playing Game System, Product Variant (Production Method), Cultural Art Traditions
316
+ ✅ 处理完成!
317
+ 保留: 9 个
318
+ 删除: 1 个
319
+ 删除的 domains: Lemur Taxon Name (Genus and Species)
320
+ ✅ 处理完成!
321
+ 保留: 10 个
322
+ 删除: 0 个
323
+ ✅ 处理完成!
324
+ 保留: 8 个
325
+ 删除: 2 个
326
+ 删除的 domains: Comic Book Series, Power Plant
327
+ ✅ 处理完成!
328
+ 保留: 9 个
329
+ 删除: 1 个
330
+ 删除的 domains: Seabird Species
331
+ ✅ 处理完成!
332
+ 保留: 8 个
333
+ 删除: 0 个
334
+ ✅ 处理完成!
335
+ 保留: 9 个
336
+ 删除: 1 个
337
+ 删除的 domains: Crop Pest
338
+
339
+ ================================================================================
340
+ 📊 合并结果...
341
+ ================================================================================
342
+
343
+ ================================================================================
344
+ 💾 保存结果...
345
+ ================================================================================
346
+
347
+ ================================================================================
348
+ 📊 统计信息:
349
+ ================================================================================
350
+ 原始 domains 数: 338
351
+ 保留 domains 数: 309
352
+ 删除 domains 数: 29
353
+ 总 attributes 数: 4370
354
+ 总耗时: 50.7 秒
355
+
356
+ 删除的 domains:
357
+ - Landmark Name
358
+ - Citrus Variety
359
+ - Tourist Attraction
360
+ - Primate Species
361
+ - Satellite Mission Type
362
+ - Fantasy Creature or Race
363
+ - Notable Sacred Sites
364
+ - Planetary Rover Name
365
+ - Commercial Aircraft Model
366
+ - Common Amphibian Species (common names)
367
+ - Mammal Species Name
368
+ - Artwork Title
369
+ - Passenger Persona (Travel Behavior)
370
+ - Solar System Planet
371
+ - Jewelry-making Technique
372
+ - Cultural Attire
373
+ - Chinese New Year Foods
374
+ - Maize (Corn) Cultivar/Variety
375
+ - Ranching Practices and Systems
376
+ - Astronomical Observatory Name
377
+ ... 还有 9 个
378
+
379
+ 💾 结果已保存到: accepted_domains_attributes_filtered.txt
380
+
381
+ ================================================================================
382
+ ✨ 处理完成!
383
+ ================================================================================
icon_generation/backup/filter_pairs.py ADDED
@@ -0,0 +1,402 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """
3
+ 使用 LLM 过滤 domain_value_pairs.txt
4
+ 移除不对应、抽象概念或标识符类型的 values
5
+ 每次处理 100 个 pairs
6
+ """
7
+
8
+ import json
9
+ import requests
10
+ from typing import List, Dict, Tuple, Optional
11
+ import time
12
+ import os
13
+
14
+
15
+ class PairFilter:
16
+ """使用 LLM 过滤 domain-value pairs"""
17
+
18
+ def __init__(self, api_key=None, base_url=None, model=None):
19
+ """初始化 LLM analyzer"""
20
+ self.api_key = api_key or os.getenv("OPENAI_API_KEY") or os.getenv("AIHUBMIX_API_KEY", "")
21
+ self.base_url = base_url or os.getenv("OPENAI_BASE_URL", "https://aihubmix.com/v1")
22
+ self.model = model or os.getenv("OPENAI_MODEL", "gemini-2.5-flash")
23
+
24
+ def filter_pairs_batch(self, pairs: List[Tuple[str, str, int]]) -> Dict:
25
+ """
26
+ 使用 LLM 批量过滤 pairs
27
+
28
+ Args:
29
+ pairs: [(domain, value, count), ...]
30
+
31
+ Returns:
32
+ {
33
+ 'keep': [(domain, value, count), ...],
34
+ 'remove': [(domain, value, count, reason), ...],
35
+ 'reasoning': 'overall explanation'
36
+ }
37
+ """
38
+ prompt = self._build_filter_prompt(pairs)
39
+
40
+ try:
41
+ response = self._query_llm(prompt)
42
+
43
+ if response:
44
+ # 清理可能的 markdown 代码块
45
+ cleaned_response = response.strip()
46
+ if cleaned_response.startswith('```'):
47
+ lines = cleaned_response.split('\n')
48
+ cleaned_response = '\n'.join(lines[1:-1] if lines[-1].strip() == '```' else lines[1:])
49
+ cleaned_response = cleaned_response.replace('```json', '').replace('```', '').strip()
50
+
51
+ result = json.loads(cleaned_response)
52
+
53
+ # 验证返回格式
54
+ if 'keep_indices' in result and 'remove_indices' in result:
55
+ keep_indices = set(result['keep_indices'])
56
+ remove_info = {item['index']: item['reason'] for item in result['remove_indices']}
57
+
58
+ keep = []
59
+ remove = []
60
+
61
+ for idx, (domain, value, count) in enumerate(pairs):
62
+ if idx in keep_indices:
63
+ keep.append((domain, value, count))
64
+ elif idx in remove_info:
65
+ remove.append((domain, value, count, remove_info[idx]))
66
+ else:
67
+ # 如果 LLM 没有明确说明,默认保留
68
+ keep.append((domain, value, count))
69
+
70
+ return {
71
+ 'keep': keep,
72
+ 'remove': remove,
73
+ 'reasoning': result.get('reasoning', '')
74
+ }
75
+ else:
76
+ print(f" ⚠️ LLM 响应缺少必需字段")
77
+ # 默认全部保留
78
+ return {
79
+ 'keep': pairs,
80
+ 'remove': [],
81
+ 'reasoning': 'LLM response format error, kept all'
82
+ }
83
+
84
+ else:
85
+ print(f" ⚠️ LLM API 调用失败")
86
+ return {
87
+ 'keep': pairs,
88
+ 'remove': [],
89
+ 'reasoning': 'LLM API failed, kept all'
90
+ }
91
+
92
+ except json.JSONDecodeError as e:
93
+ print(f" ⚠️ LLM 响应不是有效的 JSON: {e}")
94
+ print(f" 响应: {response[:300]}...")
95
+ return {
96
+ 'keep': pairs,
97
+ 'remove': [],
98
+ 'reasoning': 'JSON decode error, kept all'
99
+ }
100
+ except Exception as e:
101
+ print(f" ⚠️ 过滤错误: {e}")
102
+ return {
103
+ 'keep': pairs,
104
+ 'remove': [],
105
+ 'reasoning': f'Error: {str(e)}, kept all'
106
+ }
107
+
108
+ def _build_filter_prompt(self, pairs: List[Tuple[str, str, int]]) -> str:
109
+ """构建过滤的 prompt"""
110
+
111
+ # 构建 pairs 列表字符串
112
+ pairs_str = '\n'.join([
113
+ f" {idx}. Domain: \"{domain}\", Value: \"{value}\", Count: {count}"
114
+ for idx, (domain, value, count) in enumerate(pairs)
115
+ ])
116
+
117
+ prompt = f"""You are a data quality expert. Given a list of domain-value pairs, identify which values should be REMOVED because they:
118
+
119
+ 1. **Don't match the domain**: The value doesn't truly belong to or represent the stated domain
120
+ 2. **Are abstract concepts**: Generic, vague, or non-specific values (e.g., "Unknown", "Other", "Various")
121
+ 3. **Are mere identifiers**: Values that are just labels, codes, or IDs without semantic meaning (e.g., "TruckA", "TruckB", "Final Cost", "Option1", "Item #123")
122
+
123
+ Domain-Value Pairs to analyze:
124
+ {pairs_str}
125
+
126
+ Please analyze each pair and return ONLY the indices that should be:
127
+ - **KEPT**: Concrete, specific, meaningful values that clearly belong to their domain
128
+ - **REMOVED**: Values that fall into any of the three categories above
129
+
130
+ Return your response in the following JSON format ONLY (no markdown, no extra text):
131
+ {{
132
+ "keep_indices": [0, 2, 5, ...],
133
+ "remove_indices": [
134
+ {{"index": 1, "reason": "abstract concept"}},
135
+ {{"index": 3, "reason": "mere identifier"}},
136
+ {{"index": 4, "reason": "doesn't match domain"}},
137
+ ...
138
+ ],
139
+ "reasoning": "Brief summary of filtering approach"
140
+ }}
141
+
142
+ Examples to guide your decision:
143
+ - KEEP: "US State: California", "Sport Type: Basketball", "Industry Sector: Healthcare"
144
+ - REMOVE: "US State: Unknown" (abstract), "Vehicle: TruckA" (identifier), "Country: New York" (doesn't match - it's a city)
145
+ - REMOVE: "Category: Other", "Type: Various", "Name: Item1", "Cost: Final Cost" (all abstract/identifiers)
146
+ """
147
+
148
+ return prompt
149
+
150
+ def _query_llm(self, prompt: str) -> Optional[str]:
151
+ """查询 LLM API"""
152
+ headers = {
153
+ 'Authorization': f'Bearer {self.api_key}',
154
+ 'Content-Type': 'application/json'
155
+ }
156
+
157
+ data = {
158
+ 'model': self.model,
159
+ 'messages': [
160
+ {
161
+ 'role': 'system',
162
+ 'content': 'You are a data quality expert specialized in filtering and validating domain-value pairs. Always return valid JSON format only, without any markdown formatting or extra text.'
163
+ },
164
+ {
165
+ 'role': 'user',
166
+ 'content': prompt
167
+ }
168
+ ],
169
+ 'temperature': 0.2
170
+ }
171
+
172
+ try:
173
+ response = requests.post(
174
+ f'{self.base_url}/chat/completions',
175
+ headers=headers,
176
+ json=data,
177
+ timeout=60
178
+ )
179
+ response.raise_for_status()
180
+
181
+ result = response.json()
182
+ return result['choices'][0]['message']['content'].strip()
183
+
184
+ except requests.exceptions.Timeout:
185
+ print(" ❌ LLM API 超时")
186
+ return None
187
+ except requests.exceptions.HTTPError as e:
188
+ print(f" ❌ LLM API HTTP 错误: {e}")
189
+ return None
190
+ except requests.exceptions.RequestException as e:
191
+ print(f" ❌ LLM API 请求错误: {e}")
192
+ return None
193
+ except KeyError as e:
194
+ print(f" ❌ LLM API 响应格式错误: {e}")
195
+ return None
196
+
197
+
198
+ def load_pairs(file_path: str) -> List[Tuple[str, str, int]]:
199
+ """加载 domain_value_pairs.txt"""
200
+ pairs = []
201
+
202
+ with open(file_path, 'r', encoding='utf-8') as f:
203
+ for line in f:
204
+ line = line.strip()
205
+ if not line:
206
+ continue
207
+
208
+ # 解析 CSV 格式,处理带引号的字段
209
+ parts = []
210
+ current = []
211
+ in_quotes = False
212
+
213
+ for char in line:
214
+ if char == '"':
215
+ in_quotes = not in_quotes
216
+ elif char == ',' and not in_quotes:
217
+ parts.append(''.join(current))
218
+ current = []
219
+ else:
220
+ current.append(char)
221
+ parts.append(''.join(current))
222
+
223
+ if len(parts) >= 3:
224
+ domain = parts[0].strip()
225
+ value = parts[1].strip()
226
+ count = int(parts[2].strip())
227
+ pairs.append((domain, value, count))
228
+
229
+ return pairs
230
+
231
+
232
+ def save_filtered_pairs(pairs: List[Tuple[str, str, int]], output_file: str):
233
+ """保存过滤后的 pairs"""
234
+ with open(output_file, 'w', encoding='utf-8') as f:
235
+ for domain, value, count in pairs:
236
+ # 处理可能包含逗号的字段
237
+ if ',' in value:
238
+ value = f'"{value}"'
239
+ if ',' in domain:
240
+ domain = f'"{domain}"'
241
+ f.write(f"{domain},{value},{count}\n")
242
+
243
+
244
+ def main():
245
+ print("=" * 80)
246
+ print("Domain-Value Pairs 过滤器")
247
+ print("=" * 80)
248
+ print()
249
+
250
+ input_file = 'domain_value_pairs.txt'
251
+ output_file = 'domain_value_pairs_filtered.txt'
252
+ temp_file = 'domain_value_pairs_filtered_temp.txt'
253
+ removed_file = 'domain_value_pairs_removed.txt'
254
+
255
+ # 加载数据
256
+ print(f"📖 读取文件: {input_file}")
257
+ pairs = load_pairs(input_file)
258
+ print(f"✅ 成功读取 {len(pairs)} 个 pairs")
259
+ print()
260
+
261
+ # 检查是否有临时文件(断点续传)
262
+ filtered_pairs = []
263
+ removed_pairs = []
264
+ start_idx = 0
265
+
266
+ if os.path.exists(temp_file):
267
+ print(f"📂 发现临时文件,尝试恢复进度...")
268
+ try:
269
+ filtered_pairs = load_pairs(temp_file)
270
+ start_idx = len(filtered_pairs)
271
+ print(f"✅ 已恢复 {start_idx} 个 pairs 的处理结果")
272
+ except Exception as e:
273
+ print(f"⚠️ 临时文件读取失败: {e},从头开始")
274
+ filtered_pairs = []
275
+ start_idx = 0
276
+
277
+ if os.path.exists(removed_file):
278
+ try:
279
+ with open(removed_file, 'r', encoding='utf-8') as f:
280
+ for line in f:
281
+ if line.strip():
282
+ removed_pairs.append(line.strip())
283
+ except:
284
+ pass
285
+
286
+ # 初始化过滤器
287
+ filter_obj = PairFilter()
288
+
289
+ # 批量处理
290
+ batch_size = 100
291
+ total_batches = (len(pairs) - start_idx + batch_size - 1) // batch_size
292
+
293
+ print("=" * 80)
294
+ print(f"🚀 开始过滤处理")
295
+ print(f" 总 pairs 数: {len(pairs)}")
296
+ print(f" 已处理: {start_idx}")
297
+ print(f" 待处理: {len(pairs) - start_idx}")
298
+ print(f" 批次大小: {batch_size}")
299
+ print(f" 总批次数: {total_batches}")
300
+ print("=" * 80)
301
+ print()
302
+
303
+ start_time = time.time()
304
+
305
+ for batch_idx in range(0, len(pairs) - start_idx, batch_size):
306
+ actual_idx = start_idx + batch_idx
307
+ batch = pairs[actual_idx:actual_idx + batch_size]
308
+ current_batch_num = batch_idx // batch_size + 1
309
+
310
+ # 进度信息
311
+ progress = (actual_idx + len(batch)) / len(pairs) * 100
312
+ elapsed = time.time() - start_time
313
+ avg_time = elapsed / (batch_idx + batch_size) if batch_idx > 0 else 0
314
+ remaining = avg_time * (len(pairs) - start_idx - batch_idx - len(batch))
315
+
316
+ print(f"{'=' * 80}")
317
+ print(f"🔄 批次 {current_batch_num}/{total_batches}")
318
+ print(f" 进度: {actual_idx + len(batch)}/{len(pairs)} ({progress:.1f}%)")
319
+ print(f" 已用时间: {elapsed:.1f}秒 | 预计剩余: {remaining:.1f}秒")
320
+ print(f" 当前批次: {len(batch)} 个 pairs")
321
+
322
+ # 显示前 3 个示例
323
+ print(f" 示例:")
324
+ for i, (domain, value, count) in enumerate(batch[:3]):
325
+ print(f" {i+1}. {domain}: {value} (count: {count})")
326
+
327
+ print(f" 🤖 调用 LLM 进行过滤...")
328
+
329
+ # 调用 LLM 过滤
330
+ result = filter_obj.filter_pairs_batch(batch)
331
+
332
+ keep_count = len(result['keep'])
333
+ remove_count = len(result['remove'])
334
+
335
+ print(f" ✅ 过滤完成!")
336
+ print(f" 保留: {keep_count} 个")
337
+ print(f" 移除: {remove_count} 个")
338
+
339
+ # 显示移除的示例
340
+ if result['remove']:
341
+ print(f" 移除示例:")
342
+ for domain, value, count, reason in result['remove'][:3]:
343
+ print(f" - {domain}: {value} ({reason})")
344
+
345
+ # 更新结果
346
+ filtered_pairs.extend(result['keep'])
347
+
348
+ # 保存移除的记录
349
+ if result['remove']:
350
+ with open(removed_file, 'a', encoding='utf-8') as f:
351
+ for domain, value, count, reason in result['remove']:
352
+ if ',' in value:
353
+ value = f'"{value}"'
354
+ if ',' in domain:
355
+ domain = f'"{domain}"'
356
+ f.write(f"{domain},{value},{count},{reason}\n")
357
+ removed_pairs.extend(result['remove'])
358
+
359
+ # 保存临时文件
360
+ save_filtered_pairs(filtered_pairs, temp_file)
361
+ print(f" 💾 已保存临时结果")
362
+
363
+ # 保存最终结果
364
+ print(f"\n{'=' * 80}")
365
+ print("💾 保存最终结果...")
366
+ save_filtered_pairs(filtered_pairs, output_file)
367
+
368
+ # 删除临时文件
369
+ if os.path.exists(temp_file):
370
+ os.remove(temp_file)
371
+
372
+ elapsed_time = time.time() - start_time
373
+
374
+ # 统计信息
375
+ print(f"\n{'=' * 80}")
376
+ print("📊 过滤完成!统计信息:")
377
+ print("=" * 80)
378
+ print(f"✅ 保留 pairs: {len(filtered_pairs)} 个")
379
+ print(f"❌ 移除 pairs: {len(removed_pairs)} 个")
380
+ print(f"📋 原始 pairs: {len(pairs)} 个")
381
+ print(f"📉 过滤比例: {len(removed_pairs)/len(pairs)*100:.1f}%")
382
+ print(f"⏱️ 总耗时: {elapsed_time:.1f} 秒")
383
+
384
+ print(f"\n💾 文件保存:")
385
+ print(f" 保留的 pairs: {output_file}")
386
+ print(f" 移除的 pairs: {removed_file}")
387
+
388
+ # 显示前 10 个保留的
389
+ print(f"\n🏆 Top 10 保留的 pairs:")
390
+ for i, (domain, value, count) in enumerate(filtered_pairs[:10], 1):
391
+ print(f" {i:2d}. {domain}: {value} ({count:,})")
392
+
393
+ print(f"\n{'=' * 80}")
394
+ print("✨ 过滤完成!")
395
+ print("=" * 80)
396
+
397
+
398
+ if __name__ == '__main__':
399
+ main()
400
+
401
+
402
+
icon_generation/backup/filtered.json ADDED
The diff for this file is too large to render. See raw diff
 
icon_generation/backup/image_batch_generator.py ADDED
@@ -0,0 +1,571 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ # -*- coding: utf-8 -*-
3
+ """
4
+ 图像批量生成Pipeline
5
+ 根据topic_style.json配置,批量生成适合infographic装饰的图像
6
+ """
7
+
8
+ import os
9
+ import json
10
+ import random
11
+ import sys
12
+ import time
13
+ from typing import Dict, List, Tuple
14
+ from openai import OpenAI
15
+ from concurrent.futures import ThreadPoolExecutor
16
+ from google import genai
17
+ from google.genai import types
18
+ from PIL import Image, ImageDraw
19
+ from io import BytesIO
20
+ import numpy as np
21
+ from collections import Counter
22
+
23
+ # 添加项目根目录到路径
24
+ # sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
25
+ # from config import api_key, base_url
26
+
27
+ api_key = 'xxx'
28
+ base_url = "https://aihubmix.com/v1"
29
+
30
+ class ImageBatchGenerator:
31
+ def __init__(self):
32
+ """初始化生成器"""
33
+ # OpenAI client for text generation
34
+ self.openai_client = OpenAI(
35
+ api_key=api_key,
36
+ base_url=base_url,
37
+ )
38
+
39
+ # Gemini client for image generation
40
+ self.genai_client = genai.Client(
41
+ api_key=api_key,
42
+ http_options={"base_url": "https://aihubmix.com/gemini"},
43
+ )
44
+
45
+ # 加载topic_style配置
46
+ self.config_path = os.path.join(
47
+ os.path.dirname(os.path.dirname(os.path.abspath(__file__))),
48
+ 'generator', 'topic_style.json'
49
+ )
50
+ self.load_config()
51
+
52
+ # 输出目录
53
+ self.output_dir = os.path.join(
54
+ os.path.dirname(os.path.dirname(os.path.abspath(__file__))),
55
+ 'gen_output'
56
+ )
57
+ os.makedirs(self.output_dir, exist_ok=True)
58
+
59
+ # 设计prompt模板
60
+ self.design_prompt_template = """
61
+ [TASK START]
62
+ OBJECTIVE: Generate a text-to-image prompt for a single, isolated clipart icon based on the provided inputs.
63
+
64
+ INPUTS:
65
+ Topic: {topic}
66
+ Style Keyword: {style_keyword}
67
+ Concept: {concept}
68
+
69
+ PROCESS:
70
+ Write a text-to-image prompt describing this concept, rendered using the specified Style Keyword.
71
+
72
+ CONSTRAINTS:
73
+ - The output must be a single icon or a small, unified group of objects
74
+ - The icon MUST be isolated on a pure white background (#FFFFFF)
75
+ - No shadows, textures or patterns in the background
76
+ - The background must be completely clean and empty
77
+ - The final prompt must be concise and descriptive
78
+
79
+ REQUIRED OUTPUT:
80
+ [The final text-to-image prompt, make sure to specify "on pure white background" in the prompt]
81
+
82
+ [TASK END]
83
+ """
84
+
85
+ # 概念生成prompt
86
+ self.concept_generation_prompt = """
87
+ Generate 10 different concrete concepts for the topic "{topic}".
88
+
89
+ Requirements:
90
+ 1. Each concept must be a specific, tangible object or clear visual scene
91
+ 2. Use detailed descriptions (e.g. "stethoscope on medical chart" vs "medical")
92
+ 3. Focus on real-world items, tools, places or situations
93
+ 4. Each concept should be immediately recognizable and relatable
94
+ 5. Concepts should work well as simple icons or decorative elements
95
+ 6. Keep descriptions concise but specific
96
+
97
+ Return in this format:
98
+ 1. [concept1]
99
+ 2. [concept2]
100
+ 3. [concept3]
101
+ ...
102
+ 10. [concept10]
103
+ """
104
+
105
+ # 设计评判prompt
106
+ self.design_evaluation_prompt = """
107
+ Evaluate the following design concepts and select the 5 best ones for infographic decoration.
108
+
109
+ Evaluation criteria:
110
+ 1. Visual clarity: Easy to recognize and understand
111
+ 2. Decorative value: Suitable as decorative elements without interfering with main information
112
+ 3. Universality: Broad applicability
113
+
114
+ Design concept list:
115
+ {concepts}
116
+
117
+ Select the 5 best concepts and return in this format:
118
+ Selected concepts:
119
+ 1. [concept name]
120
+ 2. [concept name]
121
+ 3. [concept name]
122
+ 4. [concept name]
123
+ 5. [concept name]
124
+ """
125
+
126
+ def load_config(self):
127
+ """加载topic_style配置文件"""
128
+ with open(self.config_path, 'r', encoding='utf-8') as f:
129
+ self.config = json.load(f)
130
+ print(f"✅ 加载配置: {len(self.config)} 个风格类别")
131
+
132
+ def select_random_category_and_elements(self) -> Tuple[str, str, str]:
133
+ """随机选择category、keyword和topic"""
134
+ category = random.choice(list(self.config.keys()))
135
+ category_data = self.config[category]
136
+ keyword = random.choice(category_data['keywords'])
137
+ topic = random.choice(category_data['topics'])
138
+
139
+ print(f"🎯 选中: {category} | {keyword} | {topic}")
140
+ return category, keyword, topic
141
+
142
+ def generate_concepts(self, topic: str) -> List[str]:
143
+ """使用ChatGPT生成10个概念"""
144
+ print(f"🧠 生成概念...")
145
+
146
+ response = self.openai_client.chat.completions.create(
147
+ model="gpt-5-mini",
148
+ messages=[
149
+ {"role": "user", "content": self.concept_generation_prompt.format(topic=topic)}
150
+ ],
151
+ temperature=0.8
152
+ )
153
+
154
+ content = response.choices[0].message.content
155
+
156
+ # 解析概念列表 - 修复方括号解析问题
157
+ concepts = []
158
+ lines = content.strip().split('\n')
159
+ for line in lines:
160
+ line = line.strip()
161
+ if line and (line[0].isdigit() or line.startswith('-')):
162
+ # 提取方括号内的内容
163
+ if '[' in line and ']' in line:
164
+ start = line.find('[')
165
+ end = line.find(']')
166
+ if start != -1 and end != -1 and end > start:
167
+ concept = line[start+1:end].strip()
168
+ if concept:
169
+ concepts.append(concept)
170
+ else:
171
+ # 如果没有方括号,提取序号后的内容
172
+ concept = line.split('.', 1)[-1].strip()
173
+ if concept:
174
+ concepts.append(concept)
175
+
176
+ print(f"✅ 生成 {len(concepts)} 个概念")
177
+ return concepts[:10]
178
+
179
+ def evaluate_and_select_concepts(self, concepts: List[str]) -> List[str]:
180
+ """评判并选择5个最佳概念"""
181
+ print(f"🔍 评判概念...")
182
+
183
+ concepts_text = ""
184
+ for i, concept in enumerate(concepts, 1):
185
+ concepts_text += f"{i}. {concept}\n"
186
+
187
+ response = self.openai_client.chat.completions.create(
188
+ model="gpt-5-mini",
189
+ messages=[
190
+ {"role": "user", "content": self.design_evaluation_prompt.format(concepts=concepts_text)}
191
+ ],
192
+ temperature=0.3
193
+ )
194
+
195
+ content = response.choices[0].message.content
196
+
197
+ # 解析选中的概念
198
+ selected_concepts = []
199
+ lines = content.strip().split('\n')
200
+
201
+ for line in lines:
202
+ line = line.strip()
203
+ if line and line[0].isdigit() and '.' in line:
204
+ concept_name = line.split('.', 1)[1].strip()
205
+ # 在原始概念中查找匹配
206
+ for concept in concepts:
207
+ if concept_name.lower() in concept.lower() or concept.lower() in concept_name.lower():
208
+ if concept not in selected_concepts:
209
+ selected_concepts.append(concept)
210
+ break
211
+
212
+ # 如果解析不足5个,随机补充
213
+ if len(selected_concepts) < 5:
214
+ remaining = [c for c in concepts if c not in selected_concepts]
215
+ selected_concepts.extend(random.sample(remaining, min(5 - len(selected_concepts), len(remaining))))
216
+
217
+ print(f"✅ 选中 {len(selected_concepts[:5])} 个概念")
218
+ return selected_concepts[:5]
219
+
220
+ def detect_background_color(self, image: Image.Image) -> tuple:
221
+ """检测图像的背景颜色,返回(背景色, 是否为杂乱背景)"""
222
+ # 获取图像尺寸
223
+ width, height = image.size
224
+
225
+ # 采样边界点
226
+ sample_points = []
227
+
228
+ # 四个角
229
+ sample_points.extend([
230
+ (0, 0), (width-1, 0), (0, height-1), (width-1, height-1)
231
+ ])
232
+
233
+ # 边界中点
234
+ sample_points.extend([
235
+ (width//2, 0), (width//2, height-1), # 上下边中点
236
+ (0, height//2), (width-1, height//2) # 左右边中点
237
+ ])
238
+
239
+ # 边界线采样(每边采样10个点)
240
+ for i in range(1, 10):
241
+ ratio = i / 10.0
242
+ # 上边
243
+ sample_points.append((int(width * ratio), 0))
244
+ # 下边
245
+ sample_points.append((int(width * ratio), height-1))
246
+ # 左边
247
+ sample_points.append((0, int(height * ratio)))
248
+ # 右边
249
+ sample_points.append((width-1, int(height * ratio)))
250
+
251
+ # 获取所有采样点的颜色
252
+ colors = []
253
+ for x, y in sample_points:
254
+ if 0 <= x < width and 0 <= y < height:
255
+ pixel = image.getpixel((x, y))
256
+ if isinstance(pixel, int): # 灰度图
257
+ colors.append((pixel, pixel, pixel))
258
+ elif len(pixel) >= 3: # RGB或RGBA
259
+ colors.append(pixel[:3])
260
+
261
+ # 统计颜色众数
262
+ color_counts = Counter(colors)
263
+ if color_counts:
264
+ most_common_color, most_common_count = color_counts.most_common(1)[0]
265
+ total_samples = len(colors)
266
+
267
+ # 计算众数颜色占比
268
+ ratio = most_common_count / total_samples
269
+
270
+ # 如果众数颜色占比小于50%,认为背景杂乱
271
+ is_messy = ratio < 0.5
272
+
273
+ return most_common_color, is_messy
274
+
275
+ # 默认返回白色,非杂乱
276
+ return (255, 255, 255), False
277
+
278
+ def optimized_flood_fill_remove_background(self, image: Image.Image, bg_color: tuple, tolerance: int = 30) -> Image.Image:
279
+ """使用优化的flood fill算法从边界去除背景色"""
280
+ # 转换为RGBA模式
281
+ if image.mode != 'RGBA':
282
+ image = image.convert('RGBA')
283
+
284
+ # 转换为numpy数组
285
+ data = np.array(image, dtype=np.uint8)
286
+ height, width = data.shape[:2]
287
+
288
+ # 创建访问标记数组
289
+ visited = np.zeros((height, width), dtype=bool)
290
+
291
+ # 预计算颜色距离的平方(避免开方运算)
292
+ def color_distance_squared(c1, c2):
293
+ """计算颜色距离的平方,避免开方运算提高性能"""
294
+ return sum((int(a) - int(b)) ** 2 for a, b in zip(c1[:3], c2[:3]))
295
+
296
+ tolerance_squared = tolerance * tolerance
297
+
298
+ def is_background_color(pixel_color):
299
+ """判断是否为背景色,使用平方距离比较"""
300
+ return color_distance_squared(pixel_color[:3], bg_color) <= tolerance_squared
301
+
302
+ def optimized_flood_fill(start_x, start_y):
303
+ """优化的flood fill算法,使用栈而非递归,批量处理"""
304
+ if (start_y >= height or start_x >= width or
305
+ start_y < 0 or start_x < 0 or
306
+ visited[start_y, start_x]):
307
+ return
308
+
309
+ # 使用deque作为栈,性能更好
310
+ from collections import deque
311
+ stack = deque([(start_x, start_y)])
312
+ pixels_to_clear = []
313
+
314
+ while stack:
315
+ x, y = stack.pop()
316
+
317
+ # 边界检查
318
+ if x < 0 or x >= width or y < 0 or y >= height or visited[y, x]:
319
+ continue
320
+
321
+ current_color = data[y, x]
322
+
323
+ # 检查颜色是否在容差范围内
324
+ if not is_background_color(current_color):
325
+ continue
326
+
327
+ # 标记为已访问
328
+ visited[y, x] = True
329
+ pixels_to_clear.append((x, y))
330
+
331
+ # 添加相邻像素到栈中(4连通)
332
+ stack.extend([
333
+ (x+1, y), (x-1, y), (x, y+1), (x, y-1)
334
+ ])
335
+
336
+ # 批量设置像素为透明
337
+ for x, y in pixels_to_clear:
338
+ data[y, x] = (0, 0, 0, 0)
339
+
340
+ print(f" 🌊 优化Flood Fill处理...")
341
+
342
+ # 从边界开始flood fill,优化边界遍历
343
+ # 上边和下边
344
+ for x in range(0, width, 2): # 每隔一个像素采样,提高性能
345
+ optimized_flood_fill(x, 0)
346
+ optimized_flood_fill(x, height-1)
347
+
348
+ # 左边和右边
349
+ for y in range(0, height, 2): # 每隔一个像素采样,提高性能
350
+ optimized_flood_fill(0, y)
351
+ optimized_flood_fill(width-1, y)
352
+
353
+ # 补充处理边界的奇数位置
354
+ for x in range(1, width, 2):
355
+ if not visited[0, x]:
356
+ optimized_flood_fill(x, 0)
357
+ if not visited[height-1, x]:
358
+ optimized_flood_fill(x, height-1)
359
+
360
+ for y in range(1, height, 2):
361
+ if not visited[y, 0]:
362
+ optimized_flood_fill(0, y)
363
+ if not visited[y, width-1]:
364
+ optimized_flood_fill(width-1, y)
365
+
366
+ # 转换回PIL图像
367
+ return Image.fromarray(data, 'RGBA')
368
+
369
+ def crop_transparent_borders(self, image: Image.Image) -> Image.Image:
370
+ """裁剪透明边界,去除多余区域"""
371
+ if image.mode != 'RGBA':
372
+ return image
373
+
374
+ # 转换为numpy数组
375
+ data = np.array(image)
376
+
377
+ # 获取alpha通道
378
+ alpha = data[:, :, 3]
379
+
380
+ # 找到非透明像素的边界
381
+ non_transparent = np.where(alpha > 0)
382
+
383
+ if len(non_transparent[0]) == 0:
384
+ # 如果图像完全透明,返回最小尺寸
385
+ return image.crop((0, 0, 1, 1))
386
+
387
+ # 计算边界框
388
+ min_y, max_y = non_transparent[0].min(), non_transparent[0].max()
389
+ min_x, max_x = non_transparent[1].min(), non_transparent[1].max()
390
+
391
+ # 添加小的边距(5像素)
392
+ padding = 5
393
+ width, height = image.size
394
+
395
+ min_x = max(0, min_x - padding)
396
+ min_y = max(0, min_y - padding)
397
+ max_x = min(width - 1, max_x + padding)
398
+ max_y = min(height - 1, max_y + padding)
399
+
400
+ # 裁剪图像
401
+ cropped = image.crop((min_x, min_y, max_x + 1, max_y + 1))
402
+
403
+ return cropped
404
+
405
+ def post_process_image(self, image: Image.Image) -> Image.Image:
406
+ """后处理图像:去除背景并裁剪多余区域,如果背景杂乱则返回None"""
407
+ print(f" 🔧 后处理图像...")
408
+
409
+ # 检测背景颜色和杂乱程度
410
+ bg_color, is_messy = self.detect_background_color(image)
411
+
412
+ if is_messy:
413
+ print(f" ❌ 检测到杂乱背景,抛弃此图片")
414
+ return None
415
+
416
+ print(f" 📊 检测到背景色: {bg_color}")
417
+
418
+ # 使用优���的flood fill去除背景
419
+ processed_image = self.optimized_flood_fill_remove_background(image, bg_color, tolerance=30)
420
+
421
+ # 裁剪透明边界
422
+ cropped_image = self.crop_transparent_borders(processed_image)
423
+
424
+ original_size = image.size
425
+ final_size = cropped_image.size
426
+ print(f" ✂️ 尺寸调整: {original_size} → {final_size}")
427
+
428
+ return cropped_image
429
+
430
+ def generate_prompt_and_image(self, concept: str, topic: str, keyword: str, category: str) -> str:
431
+ """为单个概念生成prompt并生成图像"""
432
+ print(f" 🎨 处理: {concept[:50]}...")
433
+
434
+ # 生成设计prompt
435
+ prompt = self.design_prompt_template.format(
436
+ topic=topic,
437
+ style_keyword=keyword,
438
+ concept=concept
439
+ )
440
+
441
+ response = self.openai_client.chat.completions.create(
442
+ model="gpt-5-mini",
443
+ messages=[
444
+ {"role": "user", "content": prompt}
445
+ ],
446
+ temperature=0.7
447
+ )
448
+
449
+ image_prompt = response.choices[0].message.content.strip()
450
+
451
+ # 生成图像使用imagen-4.0,带重试机制
452
+ max_retries = 5
453
+ retry_delay = 5 # 秒
454
+ response = None
455
+
456
+ for attempt in range(max_retries):
457
+ try:
458
+ print(f" 🖼️ 生成图像 (尝试 {attempt + 1}/{max_retries})...")
459
+ response = self.genai_client.models.generate_images(
460
+ model='imagen-4.0-fast-generate-001',
461
+ prompt=image_prompt,
462
+ config=types.GenerateImagesConfig(
463
+ number_of_images=1,
464
+ aspect_ratio="1:1",
465
+ )
466
+ )
467
+ # 如果成功,跳出重试循环
468
+ if response and hasattr(response, 'generated_images') and response.generated_images:
469
+ print(f" ✅ 图像生成成功")
470
+ break
471
+ else:
472
+ print(f" ⚠️ 图像生成返回空结果")
473
+ if attempt < max_retries - 1:
474
+ print(f" ⏳ 等待 {retry_delay} 秒后重试...")
475
+ time.sleep(retry_delay)
476
+
477
+ except Exception as e:
478
+ print(f" ❌ 图像生成失败 (尝试 {attempt + 1}/{max_retries}): {str(e)}")
479
+ if attempt < max_retries - 1:
480
+ print(f" ⏳ 等待 {retry_delay} 秒后重试...")
481
+ time.sleep(retry_delay)
482
+ else:
483
+ print(f" 💀 所有重试均失败,放弃生成此图像")
484
+ return None
485
+
486
+ # 保存图像
487
+ if response and hasattr(response, 'generated_images') and response.generated_images:
488
+ generated_image = response.generated_images[0]
489
+ image = Image.open(BytesIO(generated_image.image.image_bytes))
490
+
491
+ # 后处理图像:去除背景
492
+ processed_image = self.post_process_image(image)
493
+
494
+ # 如果图像被抛弃(杂乱背景),返回None
495
+ if processed_image is None:
496
+ print(f" 🗑️ 图片已抛弃")
497
+ return None
498
+
499
+ # 构建文件名 - 使用连字符连接,下划线替换空格
500
+ safe_topic = topic.replace(' ', '_')
501
+ safe_category = category.replace(' ', '_')
502
+ safe_concept = concept[:30].replace(' ', '_')
503
+
504
+ # 移除非字母数字和允许的字符
505
+ safe_topic = "".join(c for c in safe_topic if c.isalnum() or c in ('_', '-')).strip('_-')
506
+ safe_category = "".join(c for c in safe_category if c.isalnum() or c in ('_', '-')).strip('_-')
507
+ safe_concept = "".join(c for c in safe_concept if c.isalnum() or c in ('_', '-')).strip('_-')
508
+
509
+ timestamp = int(time.time())
510
+ filename = f"{safe_topic}-{safe_category}-{safe_concept}-{timestamp}.png"
511
+ filepath = os.path.join(self.output_dir, filename)
512
+
513
+ processed_image.save(filepath)
514
+ print(f" ✅ 保存: {os.path.basename(filepath)}")
515
+ return filepath
516
+
517
+ return None
518
+
519
+ def run_pipeline(self) -> Dict:
520
+ """运行完整的批量生成pipeline"""
521
+ print("🚀 开始图像批量生成Pipeline")
522
+
523
+ # 1. 随机选择category、keyword和topic
524
+ category, keyword, topic = self.select_random_category_and_elements()
525
+
526
+ # 2. 生成10个概念
527
+ concepts = self.generate_concepts(topic)
528
+
529
+ # 3. 评判并选择5个最佳概念
530
+ selected_concepts = self.evaluate_and_select_concepts(concepts)
531
+
532
+ # 4. 并行生成prompt和图像
533
+ print(f"🖼️ 并行生成 {len(selected_concepts)} 张图像...")
534
+ generated_files = []
535
+
536
+ with ThreadPoolExecutor(max_workers=3) as executor:
537
+ futures = []
538
+ for concept in selected_concepts:
539
+ future = executor.submit(
540
+ self.generate_prompt_and_image,
541
+ concept, topic, keyword, category
542
+ )
543
+ futures.append(future)
544
+
545
+ for future in futures:
546
+ result = future.result()
547
+ if result: # 只有成功生成且未被抛弃的图片才会被添加
548
+ generated_files.append(result)
549
+
550
+ print(f"✅ 完成! 生成 {len(generated_files)} 张图像")
551
+
552
+ return {
553
+ 'category': category,
554
+ 'keyword': keyword,
555
+ 'topic': topic,
556
+ 'generated_images': len(generated_files),
557
+ 'output_files': generated_files
558
+ }
559
+
560
+
561
+ def main():
562
+ """主函数"""
563
+ random.seed(int(time.time()))
564
+ generator = ImageBatchGenerator()
565
+ for i in range(1000):
566
+ result = generator.run_pipeline()
567
+ print(f"📊 结果: {result['generated_images']} 张图像已保存")
568
+
569
+
570
+ if __name__ == "__main__":
571
+ main()
icon_generation/backup/image_batch_generator_backup.py ADDED
@@ -0,0 +1,511 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ # -*- coding: utf-8 -*-
3
+ """
4
+ 图像批量生成Pipeline
5
+ 根据topic_style.json配置,批量生成适合infographic装饰的图像
6
+ """
7
+
8
+ import os
9
+ import json
10
+ import random
11
+ import sys
12
+ import time
13
+ from typing import Dict, List, Tuple
14
+ from openai import OpenAI
15
+ from concurrent.futures import ThreadPoolExecutor
16
+ from google import genai
17
+ from google.genai import types
18
+ from PIL import Image, ImageDraw
19
+ from io import BytesIO
20
+ import numpy as np
21
+ from collections import Counter
22
+
23
+ # 添加项目根目录到路径
24
+ sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
25
+ from config import api_key, base_url
26
+
27
+ class ImageBatchGenerator:
28
+ def __init__(self):
29
+ """初始化生成器"""
30
+ # OpenAI client for text generation
31
+ self.openai_client = OpenAI(
32
+ api_key=api_key,
33
+ base_url=base_url,
34
+ )
35
+
36
+ # Gemini client for image generation
37
+ self.genai_client = genai.Client(
38
+ api_key=api_key,
39
+ http_options={"base_url": "https://aihubmix.com/gemini"},
40
+ )
41
+
42
+ # 加载topic_style配置
43
+ self.config_path = os.path.join(
44
+ os.path.dirname(os.path.dirname(os.path.abspath(__file__))),
45
+ 'generator', 'topic_style.json'
46
+ )
47
+ self.load_config()
48
+
49
+ # 输出目录
50
+ self.output_dir = os.path.join(
51
+ os.path.dirname(os.path.dirname(os.path.abspath(__file__))),
52
+ 'gen_output'
53
+ )
54
+ os.makedirs(self.output_dir, exist_ok=True)
55
+
56
+ # 设计prompt模板
57
+ self.design_prompt_template = """
58
+ [TASK START]
59
+ OBJECTIVE: Generate a text-to-image prompt for a single, isolated clipart icon based on the provided inputs.
60
+
61
+ INPUTS:
62
+ Topic: {topic}
63
+ Style Keyword: {style_keyword}
64
+ Concept: {concept}
65
+
66
+ PROCESS:
67
+ Write a text-to-image prompt describing this concept, rendered using the specified Style Keyword.
68
+
69
+ CONSTRAINTS:
70
+ - The output must be a single icon or a small, unified group of objects
71
+ - The icon MUST be isolated on a pure white background (#FFFFFF)
72
+ - No shadows, textures or patterns in the background
73
+ - The background must be completely clean and empty
74
+ - The final prompt must be concise and descriptive
75
+
76
+ REQUIRED OUTPUT:
77
+ [The final text-to-image prompt, make sure to specify "on pure white background" in the prompt]
78
+
79
+ [TASK END]
80
+ """
81
+
82
+ # 概念生成prompt
83
+ self.concept_generation_prompt = """
84
+ Generate 10 different concrete concepts for the topic "{topic}".
85
+
86
+ Requirements:
87
+ 1. Each concept must be a specific, tangible object or clear visual scene
88
+ 2. Use detailed descriptions (e.g. "stethoscope on medical chart" vs "medical")
89
+ 3. Focus on real-world items, tools, places or situations
90
+ 4. Each concept should be immediately recognizable and relatable
91
+ 5. Concepts should work well as simple icons or decorative elements
92
+ 6. Keep descriptions concise but specific
93
+
94
+ Return in this format:
95
+ 1. [concept1]
96
+ 2. [concept2]
97
+ 3. [concept3]
98
+ ...
99
+ 10. [concept10]
100
+ """
101
+
102
+ # 设计评判prompt
103
+ self.design_evaluation_prompt = """
104
+ Evaluate the following design concepts and select the 5 best ones for infographic decoration.
105
+
106
+ Evaluation criteria:
107
+ 1. Visual clarity: Easy to recognize and understand
108
+ 2. Decorative value: Suitable as decorative elements without interfering with main information
109
+ 3. Universality: Broad applicability
110
+
111
+ Design concept list:
112
+ {concepts}
113
+
114
+ Select the 5 best concepts and return in this format:
115
+ Selected concepts:
116
+ 1. [concept name]
117
+ 2. [concept name]
118
+ 3. [concept name]
119
+ 4. [concept name]
120
+ 5. [concept name]
121
+ """
122
+
123
+ def load_config(self):
124
+ """加载topic_style配置文件"""
125
+ with open(self.config_path, 'r', encoding='utf-8') as f:
126
+ self.config = json.load(f)
127
+ print(f"✅ 加载配置: {len(self.config)} 个风格类别")
128
+
129
+ def select_random_category_and_elements(self) -> Tuple[str, str, str]:
130
+ """随机选择category、keyword和topic"""
131
+ category = random.choice(list(self.config.keys()))
132
+ category_data = self.config[category]
133
+ keyword = random.choice(category_data['keywords'])
134
+ topic = random.choice(category_data['topics'])
135
+
136
+ print(f"🎯 选中: {category} | {keyword} | {topic}")
137
+ return category, keyword, topic
138
+
139
+ def generate_concepts(self, topic: str) -> List[str]:
140
+ """使用ChatGPT生成10个概念"""
141
+ print(f"🧠 生成概念...")
142
+
143
+ response = self.openai_client.chat.completions.create(
144
+ model="gpt-5-mini",
145
+ messages=[
146
+ {"role": "user", "content": self.concept_generation_prompt.format(topic=topic)}
147
+ ],
148
+ temperature=0.8
149
+ )
150
+
151
+ content = response.choices[0].message.content
152
+
153
+ # 解析概念列表 - 修复方括号解析问题
154
+ concepts = []
155
+ lines = content.strip().split('\n')
156
+ for line in lines:
157
+ line = line.strip()
158
+ if line and (line[0].isdigit() or line.startswith('-')):
159
+ # 提取方括号内的内容
160
+ if '[' in line and ']' in line:
161
+ start = line.find('[')
162
+ end = line.find(']')
163
+ if start != -1 and end != -1 and end > start:
164
+ concept = line[start+1:end].strip()
165
+ if concept:
166
+ concepts.append(concept)
167
+ else:
168
+ # 如果没有方括号,提取序号后的内容
169
+ concept = line.split('.', 1)[-1].strip()
170
+ if concept:
171
+ concepts.append(concept)
172
+
173
+ print(f"✅ 生成 {len(concepts)} 个概念")
174
+ return concepts[:10]
175
+
176
+ def evaluate_and_select_concepts(self, concepts: List[str]) -> List[str]:
177
+ """评判并选择5个最佳概念"""
178
+ print(f"🔍 评判概念...")
179
+
180
+ concepts_text = ""
181
+ for i, concept in enumerate(concepts, 1):
182
+ concepts_text += f"{i}. {concept}\n"
183
+
184
+ response = self.openai_client.chat.completions.create(
185
+ model="gpt-5-mini",
186
+ messages=[
187
+ {"role": "user", "content": self.design_evaluation_prompt.format(concepts=concepts_text)}
188
+ ],
189
+ temperature=0.3
190
+ )
191
+
192
+ content = response.choices[0].message.content
193
+
194
+ # 解析选中的概念
195
+ selected_concepts = []
196
+ lines = content.strip().split('\n')
197
+
198
+ for line in lines:
199
+ line = line.strip()
200
+ if line and line[0].isdigit() and '.' in line:
201
+ concept_name = line.split('.', 1)[1].strip()
202
+ # 在原始概念中查找匹配
203
+ for concept in concepts:
204
+ if concept_name.lower() in concept.lower() or concept.lower() in concept_name.lower():
205
+ if concept not in selected_concepts:
206
+ selected_concepts.append(concept)
207
+ break
208
+
209
+ # 如果解析不足5个,随机补充
210
+ if len(selected_concepts) < 5:
211
+ remaining = [c for c in concepts if c not in selected_concepts]
212
+ selected_concepts.extend(random.sample(remaining, min(5 - len(selected_concepts), len(remaining))))
213
+
214
+ print(f"✅ 选中 {len(selected_concepts[:5])} 个概念")
215
+ return selected_concepts[:5]
216
+
217
+ def detect_background_color(self, image: Image.Image) -> tuple:
218
+ """检测图像的背景颜色,返回(背景色, 是否为杂乱背景)"""
219
+ # 获取图像尺寸
220
+ width, height = image.size
221
+
222
+ # 采样边界点
223
+ sample_points = []
224
+
225
+ # 四个角
226
+ sample_points.extend([
227
+ (0, 0), (width-1, 0), (0, height-1), (width-1, height-1)
228
+ ])
229
+
230
+ # 边界中点
231
+ sample_points.extend([
232
+ (width//2, 0), (width//2, height-1), # 上下边中点
233
+ (0, height//2), (width-1, height//2) # 左右边中点
234
+ ])
235
+
236
+ # 边界线采样(每边采样10个点)
237
+ for i in range(1, 10):
238
+ ratio = i / 10.0
239
+ # 上边
240
+ sample_points.append((int(width * ratio), 0))
241
+ # 下边
242
+ sample_points.append((int(width * ratio), height-1))
243
+ # 左边
244
+ sample_points.append((0, int(height * ratio)))
245
+ # 右边
246
+ sample_points.append((width-1, int(height * ratio)))
247
+
248
+ # 获取所有采样点的颜色
249
+ colors = []
250
+ for x, y in sample_points:
251
+ if 0 <= x < width and 0 <= y < height:
252
+ pixel = image.getpixel((x, y))
253
+ if isinstance(pixel, int): # 灰度图
254
+ colors.append((pixel, pixel, pixel))
255
+ elif len(pixel) >= 3: # RGB或RGBA
256
+ colors.append(pixel[:3])
257
+
258
+ # 统计颜色众数
259
+ color_counts = Counter(colors)
260
+ if color_counts:
261
+ most_common_color, most_common_count = color_counts.most_common(1)[0]
262
+ total_samples = len(colors)
263
+
264
+ # 计算众数颜色占比
265
+ ratio = most_common_count / total_samples
266
+
267
+ # 如果众数颜色占比小于50%,认为背景杂乱
268
+ is_messy = ratio < 0.5
269
+
270
+ return most_common_color, is_messy
271
+
272
+ # 默认返回白色,非杂乱
273
+ return (255, 255, 255), False
274
+
275
+ def line_scan_remove_background(self, image: Image.Image, bg_color: tuple, tolerance: int = 30, min_consecutive: int = 5) -> Image.Image:
276
+ """逐行扫描去除连续的背景色区域"""
277
+ # 转换为RGBA模式
278
+ if image.mode != 'RGBA':
279
+ image = image.convert('RGBA')
280
+
281
+ # 转换为numpy数组
282
+ data = np.array(image)
283
+ width, height = image.size
284
+
285
+ def color_distance(c1, c2):
286
+ """计算颜色距离"""
287
+ return np.sqrt(sum((a - b) ** 2 for a, b in zip(c1[:3], c2[:3])))
288
+
289
+ def is_background_color(pixel_color):
290
+ """判断是否为背景色"""
291
+ return color_distance(pixel_color[:3], bg_color) <= tolerance
292
+
293
+ def process_line(line_data, is_horizontal=True):
294
+ """处理一行或一列的数据,去除连续的背景色区域"""
295
+ line_length = len(line_data)
296
+ i = 0
297
+
298
+ while i < line_length:
299
+ # 检查当前像素是否为背景色
300
+ if is_background_color(line_data[i]):
301
+ # 找到连续背景色区域的结束位置
302
+ consecutive_start = i
303
+ while i < line_length and is_background_color(line_data[i]):
304
+ i += 1
305
+ consecutive_end = i
306
+ consecutive_length = consecutive_end - consecutive_start
307
+
308
+ # 如果连续背景色区域超过阈值,设置为透明
309
+ if consecutive_length > min_consecutive:
310
+ for j in range(consecutive_start, consecutive_end):
311
+ line_data[j] = (0, 0, 0, 0) # 设置为透明
312
+ else:
313
+ i += 1
314
+
315
+ return line_data
316
+
317
+ # 逐行扫描(水平方向)
318
+ print(f" 🔍 逐行扫描(水平方向)...")
319
+ for y in range(height):
320
+ row_data = data[y, :].copy()
321
+ processed_row = process_line(row_data, is_horizontal=True)
322
+ data[y, :] = processed_row
323
+
324
+ # 逐列扫描(垂直方向)
325
+ print(f" 🔍 逐列扫描(垂直方向)...")
326
+ for x in range(width):
327
+ col_data = data[:, x].copy()
328
+ processed_col = process_line(col_data, is_horizontal=False)
329
+ data[:, x] = processed_col
330
+
331
+ # 转换回PIL图像
332
+ return Image.fromarray(data, 'RGBA')
333
+
334
+ def crop_transparent_borders(self, image: Image.Image) -> Image.Image:
335
+ """裁剪透明边界,去除多余区域"""
336
+ if image.mode != 'RGBA':
337
+ return image
338
+
339
+ # 转换为numpy数组
340
+ data = np.array(image)
341
+
342
+ # 获取alpha通道
343
+ alpha = data[:, :, 3]
344
+
345
+ # 找到非透明像素的边界
346
+ non_transparent = np.where(alpha > 0)
347
+
348
+ if len(non_transparent[0]) == 0:
349
+ # 如果图像完全透明,返回最小尺寸
350
+ return image.crop((0, 0, 1, 1))
351
+
352
+ # 计算边界框
353
+ min_y, max_y = non_transparent[0].min(), non_transparent[0].max()
354
+ min_x, max_x = non_transparent[1].min(), non_transparent[1].max()
355
+
356
+ # 添加小的边距(5像素)
357
+ padding = 5
358
+ width, height = image.size
359
+
360
+ min_x = max(0, min_x - padding)
361
+ min_y = max(0, min_y - padding)
362
+ max_x = min(width - 1, max_x + padding)
363
+ max_y = min(height - 1, max_y + padding)
364
+
365
+ # 裁剪图像
366
+ cropped = image.crop((min_x, min_y, max_x + 1, max_y + 1))
367
+
368
+ return cropped
369
+
370
+ def post_process_image(self, image: Image.Image) -> Image.Image:
371
+ """后处理图像:去除背景并裁剪多余区域,如果背景杂乱则返回None"""
372
+ print(f" 🔧 后处理图像...")
373
+
374
+ # 检测背景颜色和杂乱程度
375
+ bg_color, is_messy = self.detect_background_color(image)
376
+
377
+ if is_messy:
378
+ print(f" ❌ 检测到杂乱背景,抛弃此图片")
379
+ return None
380
+
381
+ print(f" 📊 检测到背景色: {bg_color}")
382
+
383
+ # 使用逐行扫描去除背景
384
+ processed_image = self.line_scan_remove_background(image, bg_color, tolerance=30, min_consecutive=5)
385
+
386
+ # 裁剪透明边界
387
+ cropped_image = self.crop_transparent_borders(processed_image)
388
+
389
+ original_size = image.size
390
+ final_size = cropped_image.size
391
+ print(f" ✂️ 尺寸调整: {original_size} → {final_size}")
392
+
393
+ return cropped_image
394
+
395
+ def generate_prompt_and_image(self, concept: str, topic: str, keyword: str, category: str) -> str:
396
+ """为单个概念生成prompt并生成图像"""
397
+ print(f" 🎨 处理: {concept[:50]}...")
398
+
399
+ # 生成设计prompt
400
+ prompt = self.design_prompt_template.format(
401
+ topic=topic,
402
+ style_keyword=keyword,
403
+ concept=concept
404
+ )
405
+
406
+ response = self.openai_client.chat.completions.create(
407
+ model="gpt-5-mini",
408
+ messages=[
409
+ {"role": "user", "content": prompt}
410
+ ],
411
+ temperature=0.7
412
+ )
413
+
414
+ image_prompt = response.choices[0].message.content.strip()
415
+
416
+ # 生成图像使用imagen-4.0
417
+ response = self.genai_client.models.generate_images(
418
+ model='imagen-4.0-fast-generate-001',
419
+ prompt=image_prompt,
420
+ config=types.GenerateImagesConfig(
421
+ number_of_images=1,
422
+ aspect_ratio="1:1",
423
+ )
424
+ )
425
+
426
+ # 保存图像
427
+ if response and hasattr(response, 'generated_images') and response.generated_images:
428
+ generated_image = response.generated_images[0]
429
+ image = Image.open(BytesIO(generated_image.image.image_bytes))
430
+
431
+ # 后处理图像:去除背景
432
+ processed_image = self.post_process_image(image)
433
+
434
+ # 如果图像被抛弃(杂乱背景),返回None
435
+ if processed_image is None:
436
+ print(f" 🗑️ 图片已抛弃")
437
+ return None
438
+
439
+ # 构建文件名 - 使用连字符连接,下划线替换空格
440
+ safe_topic = topic.replace(' ', '_')
441
+ safe_category = category.replace(' ', '_')
442
+ safe_concept = concept[:30].replace(' ', '_')
443
+
444
+ # 移除非字母数字和允许的字符
445
+ safe_topic = "".join(c for c in safe_topic if c.isalnum() or c in ('_', '-')).strip('_-')
446
+ safe_category = "".join(c for c in safe_category if c.isalnum() or c in ('_', '-')).strip('_-')
447
+ safe_concept = "".join(c for c in safe_concept if c.isalnum() or c in ('_', '-')).strip('_-')
448
+
449
+ timestamp = int(time.time())
450
+ filename = f"{safe_topic}-{safe_category}-{safe_concept}-{timestamp}.png"
451
+ filepath = os.path.join(self.output_dir, filename)
452
+
453
+ processed_image.save(filepath)
454
+ print(f" ✅ 保存: {os.path.basename(filepath)}")
455
+ return filepath
456
+
457
+ return None
458
+
459
+ def run_pipeline(self) -> Dict:
460
+ """运行完整的批量生成pipeline"""
461
+ print("🚀 开始图像批量生成Pipeline")
462
+
463
+ # 1. 随机选择category、keyword和topic
464
+ category, keyword, topic = self.select_random_category_and_elements()
465
+
466
+ # 2. 生成10个概念
467
+ concepts = self.generate_concepts(topic)
468
+
469
+ # 3. 评判并选择5个最佳概念
470
+ selected_concepts = self.evaluate_and_select_concepts(concepts)
471
+
472
+ # 4. 并行生成prompt和图像
473
+ print(f"🖼️ 并行生成 {len(selected_concepts)} 张图像...")
474
+ generated_files = []
475
+
476
+ with ThreadPoolExecutor(max_workers=3) as executor:
477
+ futures = []
478
+ for concept in selected_concepts:
479
+ future = executor.submit(
480
+ self.generate_prompt_and_image,
481
+ concept, topic, keyword, category
482
+ )
483
+ futures.append(future)
484
+
485
+ for future in futures:
486
+ result = future.result()
487
+ if result: # 只有成功生成且未被抛弃的图片才会被添加
488
+ generated_files.append(result)
489
+
490
+ print(f"✅ 完成! 生成 {len(generated_files)} 张图像")
491
+
492
+ return {
493
+ 'category': category,
494
+ 'keyword': keyword,
495
+ 'topic': topic,
496
+ 'generated_images': len(generated_files),
497
+ 'output_files': generated_files
498
+ }
499
+
500
+
501
+ def main():
502
+ """主函数"""
503
+ random.seed(int(time.time()))
504
+ generator = ImageBatchGenerator()
505
+ for i in range(10):
506
+ result = generator.run_pipeline()
507
+ print(f"📊 结果: {result['generated_images']} 张图像已保存")
508
+
509
+
510
+ if __name__ == "__main__":
511
+ main()
icon_generation/backup/image_generator.py ADDED
@@ -0,0 +1,64 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ from openai import OpenAI
3
+ from PIL import Image
4
+ from io import BytesIO
5
+ import base64
6
+ import sys
7
+ sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
8
+ from config import api_key, base_url
9
+
10
+ # OpenAI API configuration
11
+ API_KEY = api_key
12
+ API_PROVIDER = base_url
13
+
14
+ client = OpenAI(
15
+ api_key=API_KEY,
16
+ base_url=API_PROVIDER,
17
+ )
18
+
19
+
20
+ from openai import OpenAI
21
+ import base64
22
+ import os
23
+
24
+ client = OpenAI(
25
+ api_key=API_KEY,
26
+ base_url=API_PROVIDER
27
+ )
28
+
29
+ result = client.images.generate(
30
+ model="gpt-image-1",
31
+ prompt=prompt,
32
+ n=1, # 单次出图数量,最多 10 张
33
+ size="1024x1024", # 1024x1024 (square), 1536x1024 (3:2 landscape), 1024x1536 (2:3 portrait), auto (default)
34
+ quality="low", # high, medium, low, auto (default)
35
+ moderation="low", # low, auto (default) 需要升级 openai 包 📍
36
+ background="auto", # transparent, opaque, auto (default)
37
+ )
38
+
39
+
40
+ design_prompt_template = """
41
+ [TASK START]
42
+ OBJECTIVE: Generate a text-to-image prompt for a single, isolated clipart icon based on the provided inputs.
43
+
44
+ INPUTS:
45
+ Topic: TOPIC_PLACEHOLDER
46
+ Style Keyword: STYLE_KEYWORD_PLACEHOLDER
47
+
48
+ PROCESS:
49
+
50
+ Conceptualize: First, determine a single, specific, and universally recognizable visual concept (a concrete object or scene) that best represents the broad Topic.
51
+ Generate: Second, write a text-to-image prompt describing this visual concept, rendered using the specified Style Keyword.
52
+
53
+ CONSTRAINTS:
54
+
55
+ The output must be a single icon or a small, unified group of objects.
56
+ The icon MUST be isolated on a plain white background.
57
+ The final prompt must be concise and descriptive.
58
+
59
+ REQUIRED OUTPUT: (Provide your answer in this exact structure)
60
+ Selected Concept: [The specific visual metaphor you chose for the topic]
61
+ Image Prompt: [The final text-to-image prompt]
62
+
63
+ [TASK END]
64
+ """
icon_generation/backup/json_to_txt.py ADDED
@@ -0,0 +1,77 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """
3
+ 将 refined_domains.json 转换为 txt 格式
4
+ 每行包含: domain, specific_value, value_count
5
+ 按 value_count 降序排序
6
+ """
7
+
8
+ import json
9
+
10
+
11
+ def convert_json_to_txt(input_file: str, output_file: str):
12
+ """
13
+ 转换 JSON 到 TXT 格式
14
+
15
+ Args:
16
+ input_file: 输入的 JSON 文件
17
+ output_file: 输出的 TXT 文件
18
+ """
19
+ print(f"📖 读取文件: {input_file}")
20
+
21
+ # 读取 JSON 数据
22
+ with open(input_file, 'r', encoding='utf-8') as f:
23
+ data = json.load(f)
24
+
25
+ print(f"✅ 成功读取 {len(data)} 个 domains")
26
+
27
+ # 收集所有的 domain-value-count 三元组
28
+ all_pairs = []
29
+
30
+ for item in data:
31
+ domain = item['domain']
32
+ for value_info in item['values']:
33
+ value = value_info['value']
34
+ count = value_info['count']
35
+ all_pairs.append((domain, value, count))
36
+
37
+ print(f"📊 总共收集 {len(all_pairs)} 个 domain-value pairs")
38
+
39
+ # 按 count 降序排序
40
+ print("🔄 按 value_count 降序排序...")
41
+ all_pairs.sort(key=lambda x: x[2], reverse=True)
42
+
43
+ # 写入 TXT 文件
44
+ print(f"💾 写入文件: {output_file}")
45
+
46
+ with open(output_file, 'w', encoding='utf-8') as f:
47
+ for domain, value, count in all_pairs:
48
+ # 确保 value 中的逗号不会破坏 CSV 格式
49
+ # 如果 value 包含逗号,用引号包裹
50
+ if ',' in value:
51
+ value = f'"{value}"'
52
+ if ',' in domain:
53
+ domain = f'"{domain}"'
54
+
55
+ f.write(f"{domain},{value},{count}\n")
56
+
57
+ print(f"✅ 转换完成!")
58
+ print(f"\n📈 统计信息:")
59
+ print(f" 总 pairs 数: {len(all_pairs):,}")
60
+ print(f" 最高 count: {all_pairs[0][2]:,} ({all_pairs[0][0]}: {all_pairs[0][1]})")
61
+ print(f" 最低 count: {all_pairs[-1][2]:,} ({all_pairs[-1][0]}: {all_pairs[-1][1]})")
62
+
63
+ # 显示前 10 个
64
+ print(f"\n🏆 Top 10 pairs:")
65
+ for i, (domain, value, count) in enumerate(all_pairs[:10], 1):
66
+ print(f" {i:2d}. {domain}: {value} ({count:,})")
67
+
68
+
69
+ if __name__ == '__main__':
70
+ input_file = 'refined_domains.json'
71
+ output_file = 'domain_value_pairs.txt'
72
+
73
+ convert_json_to_txt(input_file, output_file)
74
+
75
+
76
+
77
+
icon_generation/backup/refine_domains.py ADDED
@@ -0,0 +1,537 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """
3
+ 使用 LLM 对 filtered.json 中的 name 字段进行 refine,
4
+ 将其改为更明确的 domain,并按 count 排序
5
+ """
6
+
7
+ import json
8
+ import os
9
+ import requests
10
+ from typing import Dict, Optional, List
11
+ from collections import Counter
12
+ from concurrent.futures import ThreadPoolExecutor, as_completed
13
+ import threading
14
+
15
+
16
+ class DomainRefiner:
17
+ """使用 LLM 来 refine domain names"""
18
+
19
+ def __init__(self, api_key=None, base_url=None, model=None):
20
+ """
21
+ 初始化 LLM analyzer
22
+
23
+ Args:
24
+ api_key: API key
25
+ base_url: API base URL
26
+ model: Model name
27
+ """
28
+ self.api_key = api_key or os.getenv("OPENAI_API_KEY") or os.getenv("AIHUBMIX_API_KEY", "")
29
+ self.base_url = base_url or os.getenv("OPENAI_BASE_URL", "https://aihubmix.com/v1")
30
+ self.model = model or os.getenv("OPENAI_MODEL", "gemini-2.5-flash")
31
+ self.lock = threading.Lock() # 线程锁,用于打印
32
+
33
+ def filter_values(self, domain_name: str, values_with_counts: List[Dict]) -> List[Dict]:
34
+ """
35
+ 使用 LLM 过滤掉不属于该 domain 的 specific values 和相似/重复的 values
36
+
37
+ Args:
38
+ domain_name: domain 名称
39
+ values_with_counts: 包含 value 和 count 的列表 [{"value": "...", "count": ...}, ...]
40
+
41
+ Returns:
42
+ 过滤后的 values 列表
43
+ """
44
+ # 如果值太多,只发送 top 50 给 LLM
45
+ values_to_check = values_with_counts[:50] if len(values_with_counts) > 50 else values_with_counts
46
+
47
+ prompt = self._build_filter_prompt(domain_name, values_to_check)
48
+
49
+ try:
50
+ response = self._query_llm(prompt)
51
+
52
+ if response:
53
+ # 清理可能的 markdown 代码块
54
+ cleaned_response = response.strip()
55
+ if cleaned_response.startswith('```'):
56
+ lines = cleaned_response.split('\n')
57
+ cleaned_response = '\n'.join(lines[1:-1] if lines[-1].strip() == '```' else lines[1:])
58
+ cleaned_response = cleaned_response.replace('```json', '').replace('```', '').strip()
59
+
60
+ result = json.loads(cleaned_response)
61
+
62
+ # 验证返回格式
63
+ if 'filtered_values' in result:
64
+ filtered_values_set = set(result['filtered_values'])
65
+
66
+ # 对于 top 50,过滤它们
67
+ filtered_top = [v for v in values_to_check if v['value'] in filtered_values_set]
68
+
69
+ # 如果原始列表更长,保留剩余的(因为没有检查)
70
+ if len(values_with_counts) > 50:
71
+ remaining = values_with_counts[50:]
72
+ filtered_list = filtered_top + remaining
73
+ else:
74
+ filtered_list = filtered_top
75
+
76
+ return filtered_list
77
+ else:
78
+ with self.lock:
79
+ print(f" ⚠️ LLM 过滤响应缺少字段,保留原始值")
80
+ return values_with_counts
81
+
82
+ else:
83
+ with self.lock:
84
+ print(f" ⚠️ LLM 过滤失败,保留原始值")
85
+ return values_with_counts
86
+
87
+ except json.JSONDecodeError as e:
88
+ with self.lock:
89
+ print(f" ⚠️ LLM 过滤响应不是有效的 JSON,保留原始值")
90
+ return values_with_counts
91
+ except Exception as e:
92
+ with self.lock:
93
+ print(f" ⚠️ 值过滤错误: {e},保留原始值")
94
+ return values_with_counts
95
+
96
+ def _build_filter_prompt(self, domain_name: str, values_with_counts: List[Dict]) -> str:
97
+ """构建值过滤的 prompt"""
98
+
99
+ values_str = '\n'.join([f' - "{v["value"]}" (count: {v["count"]})' for v in values_with_counts])
100
+
101
+ prompt = f"""Given a domain name and its associated values, please filter out:
102
+ 1. Values that don't truly belong to this domain (too specific, off-topic, or irrelevant)
103
+ 2. Similar or duplicate values (keep the most common or representative one)
104
+ 3. Values that are too generic or ambiguous
105
+
106
+ Domain: "{domain_name}"
107
+
108
+ Values to filter:
109
+ {values_str}
110
+
111
+ Please analyze these values and return ONLY the values that:
112
+ - Clearly belong to this domain
113
+ - Are distinct (not duplicates or very similar)
114
+ - Are meaningful attributes
115
+
116
+ Return your response in the following JSON format ONLY (no markdown, no extra text):
117
+ {{
118
+ "filtered_values": ["value1", "value2", ...],
119
+ "removed_count": number_of_removed_values,
120
+ "reasoning": "Brief explanation of filtering criteria used"
121
+ }}
122
+
123
+ Examples:
124
+ - Domain "Movie Genre" with values ["Action", "action movie", "ACT"] → Keep only "Action"
125
+ - Domain "Country" with values ["USA", "New York", "California"] → Remove "New York", "California" (cities, not countries)
126
+ """
127
+
128
+ return prompt
129
+
130
+ def refine_domain_name(self, name: str, sample_values: List[str], total_count: int) -> Dict:
131
+ """
132
+ 使用 LLM 分析并 refine domain name
133
+
134
+ Args:
135
+ name: 原始 name
136
+ sample_values: 一些示例 values
137
+ total_count: 总计数
138
+
139
+ Returns:
140
+ {
141
+ 'original_name': str,
142
+ 'refined_domain': str,
143
+ 'reasoning': str
144
+ }
145
+ """
146
+ prompt = self._build_domain_refinement_prompt(name, sample_values, total_count)
147
+
148
+ try:
149
+ response = self._query_llm(prompt)
150
+
151
+ if response:
152
+ # 清理可能的 markdown 代码块
153
+ cleaned_response = response.strip()
154
+ if cleaned_response.startswith('```'):
155
+ lines = cleaned_response.split('\n')
156
+ cleaned_response = '\n'.join(lines[1:-1] if lines[-1].strip() == '```' else lines[1:])
157
+ cleaned_response = cleaned_response.replace('```json', '').replace('```', '').strip()
158
+
159
+ result = json.loads(cleaned_response)
160
+
161
+ # 验证返回格式
162
+ if 'refined_domain' in result:
163
+ result['original_name'] = name
164
+ return result
165
+ else:
166
+ with self.lock:
167
+ print(f" ❌ LLM 响应缺少必需字段: {result}")
168
+ return None
169
+
170
+ else:
171
+ with self.lock:
172
+ print(" ❌ LLM API 调用失败")
173
+ return None
174
+
175
+ except json.JSONDecodeError as e:
176
+ with self.lock:
177
+ print(f" ❌ LLM 响应不是有效的 JSON: {e}")
178
+ print(f" 响应内容: {response[:500]}...")
179
+ return None
180
+ except Exception as e:
181
+ with self.lock:
182
+ print(f" ❌ Domain refinement 错误: {e}")
183
+ return None
184
+
185
+ def _build_domain_refinement_prompt(self, name: str, sample_values: List[str], total_count: int) -> str:
186
+ """构建 domain refinement 的 prompt"""
187
+
188
+ sample_values_str = ', '.join(f'"{v}"' for v in sample_values[:10])
189
+
190
+ prompt = f"""Given a data field name and its sample values, please refine the name to a more specific and clear domain name.
191
+
192
+ The domain name should:
193
+ 1. Clearly indicate what category/dimension this field represents
194
+ 2. Be consistent and professional
195
+ 3. Form a clear "domain: specific attribute" relationship with its values
196
+ 4. Be concise (1-3 words)
197
+
198
+ Input Information:
199
+ - Original Field Name: "{name}"
200
+ - Sample Values: {sample_values_str}
201
+ - Total Entries Count: {total_count}
202
+
203
+ Please analyze the field name and sample values, then provide:
204
+ 1. A refined domain name that better describes this dimension
205
+ 2. Brief reasoning for your choice
206
+
207
+ Return your response in the following JSON format ONLY (no markdown, no extra text):
208
+ {{
209
+ "refined_domain": "YourRefinedDomainName",
210
+ "reasoning": "Brief explanation of why this domain name is more appropriate"
211
+ }}
212
+
213
+ Examples:
214
+ - Original: "Genre" with values ["Action", "Rock", "Pop"] → Refined: "Entertainment Genre"
215
+ - Original: "Type" with values ["Movie", "TV Show"] → Refined: "Media Type"
216
+ - Original: "Category" with values ["Electronics", "Books"] → Refined: "Product Category"
217
+ """
218
+
219
+ return prompt
220
+
221
+ def _query_llm(self, prompt: str) -> Optional[str]:
222
+ """
223
+ 查询 LLM API
224
+
225
+ Args:
226
+ prompt: 发送给 LLM 的 prompt
227
+
228
+ Returns:
229
+ str: LLM 响应内容
230
+ """
231
+ headers = {
232
+ 'Authorization': f'Bearer {self.api_key}',
233
+ 'Content-Type': 'application/json'
234
+ }
235
+
236
+ data = {
237
+ 'model': self.model,
238
+ 'messages': [
239
+ {
240
+ 'role': 'system',
241
+ 'content': 'You are a data modeling expert specialized in creating clear, semantic domain names. Always return valid JSON format only, without any markdown formatting or extra text.'
242
+ },
243
+ {
244
+ 'role': 'user',
245
+ 'content': prompt
246
+ }
247
+ ],
248
+ 'temperature': 0.3
249
+ }
250
+
251
+ try:
252
+ response = requests.post(
253
+ f'{self.base_url}/chat/completions',
254
+ headers=headers,
255
+ json=data,
256
+ timeout=30
257
+ )
258
+ response.raise_for_status()
259
+
260
+ result = response.json()
261
+ return result['choices'][0]['message']['content'].strip()
262
+
263
+ except requests.exceptions.Timeout:
264
+ with self.lock:
265
+ print("❌ LLM API 超时")
266
+ return None
267
+ except requests.exceptions.HTTPError as e:
268
+ with self.lock:
269
+ print(f"❌ LLM API HTTP 错误: {e}")
270
+ if hasattr(e.response, 'text'):
271
+ print(f" 响应: {e.response.text[:200]}")
272
+ return None
273
+ except requests.exceptions.RequestException as e:
274
+ with self.lock:
275
+ print(f"❌ LLM API 请求错误: {e}")
276
+ return None
277
+ except KeyError as e:
278
+ with self.lock:
279
+ print(f"❌ LLM API 响应格式错误: {e}")
280
+ return None
281
+
282
+
283
+ def process_single_item(item: Dict, idx: int, total: int, refiner: DomainRefiner, start_time: float, start_idx: int) -> Dict:
284
+ """
285
+ 处理单个数据项
286
+
287
+ Args:
288
+ item: 数据项
289
+ idx: 当前索引
290
+ total: 总数量
291
+ refiner: DomainRefiner 实例
292
+ start_time: 开始时间
293
+ start_idx: 起始索引
294
+
295
+ Returns:
296
+ 处理后的数据项
297
+ """
298
+ import time
299
+
300
+ # 计算进度信息
301
+ progress = (idx + 1) / total * 100
302
+ elapsed = time.time() - start_time
303
+ avg_time = elapsed / (idx - start_idx + 1) if idx > start_idx else 0
304
+ remaining = avg_time * (total - idx - 1)
305
+
306
+ with refiner.lock:
307
+ print(f"\n{'=' * 80}")
308
+ print(f"🔄 处理进度: {idx + 1}/{total} ({progress:.1f}%)")
309
+ print(f"⏱️ 已用时间: {elapsed:.1f}秒 | 预计剩余: {remaining:.1f}秒")
310
+ print(f"📝 当前字段: {item['name']}")
311
+ print(f" - 值数量: {item['num_values']}")
312
+ print(f" - 总计数: {item['total_count']}")
313
+
314
+ # 提取 sample values
315
+ sample_values = [v['value'] for v in item['values'][:15]]
316
+
317
+ with refiner.lock:
318
+ print(f" - 示例值: {', '.join(sample_values[:5])}")
319
+ print(f"🤖 调用 LLM 进行 domain refinement...")
320
+
321
+ # 使用 LLM refine domain name
322
+ refined_result = refiner.refine_domain_name(
323
+ item['name'],
324
+ sample_values,
325
+ item['total_count']
326
+ )
327
+
328
+ if refined_result:
329
+ refined_domain = refined_result['refined_domain']
330
+ reasoning = refined_result.get('reasoning', '')
331
+
332
+ with refiner.lock:
333
+ print(f"✅ Domain refinement 成功!")
334
+ print(f" 原始名称: '{item['name']}'")
335
+ print(f" 优化域名: '{refined_domain}'")
336
+ print(f" 优化理由: {reasoning}")
337
+ print(f"🔍 调用 LLM 进行 value filtering...")
338
+
339
+ # 过滤 values
340
+ filtered_values = refiner.filter_values(refined_domain, item['values'])
341
+
342
+ with refiner.lock:
343
+ removed_count = len(item['values']) - len(filtered_values)
344
+ print(f"✅ Value filtering 完成!")
345
+ print(f" 原始值数量: {len(item['values'])}")
346
+ print(f" 过滤后数量: {len(filtered_values)}")
347
+ print(f" 移除数量: {removed_count}")
348
+
349
+ # 构建新的数据项
350
+ refined_item = {
351
+ 'domain': refined_domain,
352
+ 'original_name': item['name'],
353
+ 'num_values': len(filtered_values),
354
+ 'original_num_values': item['num_values'],
355
+ 'total_count': sum(v['count'] for v in filtered_values),
356
+ 'original_total_count': item['total_count'],
357
+ 'values': filtered_values,
358
+ 'refinement_reasoning': reasoning
359
+ }
360
+
361
+ return refined_item
362
+ else:
363
+ # 如果 LLM 失败,保留原始 name 作为 domain
364
+ with refiner.lock:
365
+ print(f"⚠️ LLM refinement 失败,使用原始 name,跳过值过滤")
366
+
367
+ refined_item = {
368
+ 'domain': item['name'],
369
+ 'original_name': item['name'],
370
+ 'num_values': item['num_values'],
371
+ 'original_num_values': item['num_values'],
372
+ 'total_count': item['total_count'],
373
+ 'original_total_count': item['total_count'],
374
+ 'values': item['values'],
375
+ 'refinement_reasoning': 'LLM refinement failed, kept original'
376
+ }
377
+
378
+ return refined_item
379
+
380
+
381
+ def process_filtered_json(input_file: str, output_file: str, temp_file: str = None, num_threads: int = 10):
382
+ """
383
+ 处理 filtered.json 文件(并行处理)
384
+
385
+ Args:
386
+ input_file: 输入文件路径
387
+ output_file: 输出文件路径
388
+ temp_file: 临时文件路径,用于实时保存中间结果
389
+ num_threads: 并行线程数
390
+ """
391
+ import os
392
+ import time
393
+
394
+ if temp_file is None:
395
+ temp_file = output_file.replace('.json', '_temp.json')
396
+
397
+ print(f"📖 读取文件: {input_file}")
398
+ print(f"💾 临时文件: {temp_file}")
399
+ print(f"✨ 最终文件: {output_file}")
400
+ print(f"🔧 并行线程数: {num_threads}")
401
+
402
+ # 读取原始数据
403
+ with open(input_file, 'r', encoding='utf-8') as f:
404
+ data = json.load(f)
405
+
406
+ print(f"✅ 成功读取 {len(data)} 个字段\n")
407
+ print("=" * 80)
408
+
409
+ # 检查是否已有临时文件(支持断点续传)
410
+ refined_data = []
411
+ start_idx = 0
412
+ processed_names = set()
413
+
414
+ if os.path.exists(temp_file):
415
+ print(f"📂 发现临时文件,尝试恢复进度...")
416
+ try:
417
+ with open(temp_file, 'r', encoding='utf-8') as f:
418
+ refined_data = json.load(f)
419
+ start_idx = len(refined_data)
420
+ processed_names = {item['original_name'] for item in refined_data}
421
+ print(f"✅ 已恢复 {start_idx} 个字段的处理结果")
422
+ except Exception as e:
423
+ print(f"⚠️ 临时文件读取失败: {e},从头开始")
424
+ refined_data = []
425
+ start_idx = 0
426
+ processed_names = set()
427
+
428
+ # 过滤掉已处理的项
429
+ items_to_process = [item for item in data if item['name'] not in processed_names]
430
+
431
+ if not items_to_process:
432
+ print("✅ 所有项目已处理完成!")
433
+ return
434
+
435
+ print(f"📋 待处理项目: {len(items_to_process)} 个")
436
+ print("=" * 80)
437
+
438
+ # 初始化 LLM refiner
439
+ refiner = DomainRefiner()
440
+
441
+ # 记录开始时间
442
+ start_time = time.time()
443
+
444
+ # 使用线程池并行处理
445
+ with ThreadPoolExecutor(max_workers=num_threads) as executor:
446
+ # 提交所有任务
447
+ future_to_idx = {
448
+ executor.submit(
449
+ process_single_item,
450
+ item,
451
+ start_idx + i,
452
+ len(data),
453
+ refiner,
454
+ start_time,
455
+ start_idx
456
+ ): (start_idx + i, item)
457
+ for i, item in enumerate(items_to_process)
458
+ }
459
+
460
+ # 按完成顺序收集结果
461
+ completed = 0
462
+ for future in as_completed(future_to_idx):
463
+ idx, item = future_to_idx[future]
464
+ try:
465
+ result = future.result()
466
+ refined_data.append(result)
467
+ completed += 1
468
+
469
+ # 每处理 5 个项目保存一次
470
+ if completed % 5 == 0:
471
+ with refiner.lock:
472
+ print(f"\n{'=' * 80}")
473
+ print(f"💾 保存中间结果... (已完成 {completed}/{len(items_to_process)})")
474
+ with open(temp_file, 'w', encoding='utf-8') as f:
475
+ json.dump(refined_data, f, ensure_ascii=False, indent=2)
476
+ with refiner.lock:
477
+ print(f"✅ 已保存 {len(refined_data)} 个字段到临时文件")
478
+
479
+ except Exception as e:
480
+ with refiner.lock:
481
+ print(f"\n❌ 处理项目 {idx} ({item['name']}) 时出错: {e}")
482
+
483
+ # 最终保存一次
484
+ print(f"\n{'=' * 80}")
485
+ print(f"💾 保存最终临时结果...")
486
+ with open(temp_file, 'w', encoding='utf-8') as f:
487
+ json.dump(refined_data, f, ensure_ascii=False, indent=2)
488
+
489
+ # 按 total_count 降序排序
490
+ print(f"\n{'=' * 80}")
491
+ print("📊 按 total_count 进行排序...")
492
+ refined_data.sort(key=lambda x: x['total_count'], reverse=True)
493
+
494
+ # 保存最终结果
495
+ print(f"💾 保存最终结果到: {output_file}")
496
+ with open(output_file, 'w', encoding='utf-8') as f:
497
+ json.dump(refined_data, f, ensure_ascii=False, indent=2)
498
+
499
+ print(f"\n{'=' * 80}")
500
+ print(f"✅ 处理完成!共处理 {len(refined_data)} 个字段")
501
+ print(f"⏱️ 总耗时: {time.time() - start_time:.1f}秒")
502
+
503
+ # 输出统计信息
504
+ print(f"\n{'=' * 80}")
505
+ print("📈 统计信息:")
506
+ print(f" 总字段数: {len(refined_data)}")
507
+ print(f" 总记录数(过滤后): {sum(item['total_count'] for item in refined_data):,}")
508
+ print(f" 总记录数(原始): {sum(item.get('original_total_count', item['total_count']) for item in refined_data):,}")
509
+
510
+ # 计算过滤统计
511
+ total_values_before = sum(item.get('original_num_values', item['num_values']) for item in refined_data)
512
+ total_values_after = sum(item['num_values'] for item in refined_data)
513
+ print(f" 总值数量(原始): {total_values_before:,}")
514
+ print(f" 总值数量(过滤后): {total_values_after:,}")
515
+ print(f" 过滤比例: {(1 - total_values_after/total_values_before)*100:.1f}%")
516
+
517
+ # 显示前 10 个 domain
518
+ print(f"\n{'=' * 80}")
519
+ print("🏆 Top 10 Domains (按 total_count):")
520
+ for i, item in enumerate(refined_data[:10]):
521
+ print(f"\n {i+1}. {item['domain']} (原: {item['original_name']})")
522
+ print(f" Count: {item['total_count']:,}, Values: {item['num_values']}")
523
+ print(f" 示例值: {', '.join([v['value'] for v in item['values'][:5]])}")
524
+
525
+ # 删除临时文件
526
+ if os.path.exists(temp_file):
527
+ print(f"\n🗑�� 保留临时文件以备恢复: {temp_file}")
528
+ # os.remove(temp_file) # 暂时不删除,以便需要时恢复
529
+
530
+
531
+ if __name__ == '__main__':
532
+ input_file = '/home/lizhen/ChartPipeline/icon_generation/filtered.json'
533
+ output_file = '/home/lizhen/ChartPipeline/icon_generation/refined_domains.json'
534
+
535
+ process_filtered_json(input_file, output_file, num_threads=10)
536
+
537
+
icon_generation/backup/refined_domains.json ADDED
The diff for this file is too large to render. See raw diff
 
icon_generation/backup/summarize_domains.py ADDED
@@ -0,0 +1,142 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """
3
+ 总结 domain_value_pairs_enhanced.txt 中的所有 domains 及其 top 5 attributes
4
+ """
5
+
6
+ from collections import defaultdict
7
+ from typing import List, Tuple
8
+
9
+
10
+ def parse_csv_line(line: str) -> Tuple[str, str, int]:
11
+ """解析 CSV 行"""
12
+ parts = []
13
+ current = []
14
+ in_quotes = False
15
+
16
+ for char in line:
17
+ if char == '"':
18
+ in_quotes = not in_quotes
19
+ elif char == ',' and not in_quotes:
20
+ parts.append(''.join(current).strip())
21
+ current = []
22
+ else:
23
+ current.append(char)
24
+ parts.append(''.join(current).strip())
25
+
26
+ if len(parts) >= 3:
27
+ domain = parts[0]
28
+ attribute = parts[1]
29
+ freq = int(parts[2])
30
+ return domain, attribute, freq
31
+ else:
32
+ return None, None, 0
33
+
34
+
35
+ def summarize_domains(input_file: str, output_file: str = None):
36
+ """
37
+ 总结所有 domains 及其 top 5 attributes
38
+
39
+ Args:
40
+ input_file: 输入文件路径
41
+ output_file: 输出文件路径(可选)
42
+ """
43
+ print("=" * 80)
44
+ print("Domain 和 Top Attributes 汇总")
45
+ print("=" * 80)
46
+ print()
47
+
48
+ print(f"📖 读取文件: {input_file}")
49
+
50
+ # 收集每个 domain 的所有 attributes
51
+ domain_attributes = defaultdict(list)
52
+
53
+ with open(input_file, 'r', encoding='utf-8') as f:
54
+ for line in f:
55
+ line = line.strip()
56
+ if not line:
57
+ continue
58
+
59
+ domain, attribute, freq = parse_csv_line(line)
60
+ if domain:
61
+ domain_attributes[domain].append((attribute, freq))
62
+
63
+ print(f"✅ 成功读取 {len(domain_attributes)} 个 domains")
64
+ print()
65
+
66
+ # 对每个 domain 的 attributes 按 frequency 排序
67
+ for domain in domain_attributes:
68
+ domain_attributes[domain].sort(key=lambda x: x[1], reverse=True)
69
+
70
+ # 按 domain 的 top attribute 的 frequency 排序 domains
71
+ sorted_domains = sorted(
72
+ domain_attributes.items(),
73
+ key=lambda x: x[1][0][1] if x[1] else 0,
74
+ reverse=True
75
+ )
76
+
77
+ # 生成输出
78
+ print("=" * 80)
79
+ print("📊 所有 Domains 及 Top 5 Attributes:")
80
+ print("=" * 80)
81
+ print()
82
+
83
+ summary_lines = []
84
+
85
+ for idx, (domain, attributes) in enumerate(sorted_domains, 1):
86
+ # 获取 top 5 attributes
87
+ top_5 = attributes[:5]
88
+ top_5_names = [attr for attr, _ in top_5]
89
+
90
+ # 格式化输出
91
+ line1 = f"{idx}. {domain}"
92
+ line2 = f"Attributes: {', '.join(top_5_names)}"
93
+
94
+ summary_lines.append(line1)
95
+ summary_lines.append(line2)
96
+ summary_lines.append("") # 空行
97
+
98
+ print(line1)
99
+ print(line2)
100
+ print()
101
+
102
+ # 保存到文件(如果指定)
103
+ if output_file:
104
+ print("=" * 80)
105
+ print(f"💾 保存到文件: {output_file}")
106
+
107
+ with open(output_file, 'w', encoding='utf-8') as f:
108
+ f.write("Domain Summary with Top 5 Attributes\n")
109
+ f.write("=" * 80 + "\n\n")
110
+ for line in summary_lines:
111
+ f.write(line + "\n")
112
+
113
+ print(f"✅ 已保存")
114
+
115
+ # 统计信息
116
+ print()
117
+ print("=" * 80)
118
+ print("📈 统计信息:")
119
+ print("=" * 80)
120
+ print(f"总 Domains 数: {len(domain_attributes)}")
121
+ print(f"总 Attributes 数: {sum(len(attrs) for attrs in domain_attributes.values())}")
122
+
123
+ # 计算每个 domain 的平均 attributes 数
124
+ avg_attrs = sum(len(attrs) for attrs in domain_attributes.values()) / len(domain_attributes)
125
+ print(f"平均每个 Domain 的 Attributes 数: {avg_attrs:.1f}")
126
+
127
+ # Domains 按 attributes 数量分布
128
+ attr_counts = [len(attrs) for attrs in domain_attributes.values()]
129
+ print(f"Attributes 数量范围: {min(attr_counts)} - {max(attr_counts)}")
130
+
131
+ print()
132
+ print("=" * 80)
133
+ print("✨ 汇总完成!")
134
+ print("=" * 80)
135
+
136
+
137
+ if __name__ == '__main__':
138
+ input_file = 'domain_value_pairs_enhanced.txt'
139
+ output_file = 'domain_summary.txt'
140
+
141
+ summarize_domains(input_file, output_file)
142
+
icon_generation/backup/topic_style.json ADDED
@@ -0,0 +1,520 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "flat_and_minimal": {
3
+ "description": "扁平与极简:最清晰、最现代、最通用的信息图风格,专注于形状和颜色的直接传达。",
4
+ "keywords": [
5
+ "flat_solid_icon",
6
+ "minimal_shape",
7
+ "solid_color_fill",
8
+ "negative_space_icon",
9
+ "modern_minimal_form",
10
+ "basic_geometric_shape",
11
+ "essential_silhouette",
12
+ "flat_ui_style",
13
+ "simplified_form",
14
+ "color_block_icon",
15
+ "app_style_icon",
16
+ "wayfinding_style",
17
+ "flat_badge",
18
+ "corporate_flat",
19
+ "sharp_edge_vector"
20
+ ],
21
+ "topics": [
22
+ "business information",
23
+ "economy",
24
+ "products and services",
25
+ "online and remote learning",
26
+ "school",
27
+ "sustainability",
28
+ "public health",
29
+ "government policy",
30
+ "technology and engineering",
31
+ "demographics",
32
+ "social condition",
33
+ "welfare"
34
+ ]
35
+ },
36
+ "outline_and_line_art": {
37
+ "description": "描边与线条:使用线条而非填充,营造出更轻盈、更技术感或更优雅的视觉感受。",
38
+ "keywords": [
39
+ "monoline_icon",
40
+ "outline_glyph",
41
+ "thin_stroke_style",
42
+ "bold_outline_shape",
43
+ "line_art_pictogram",
44
+ "continuous_line_icon",
45
+ "minimal_stroke_design",
46
+ "technical_drawing_style",
47
+ "wireframe_style_icon",
48
+ "stroked_vector_shape",
49
+ "rounded_outline_icon",
50
+ "sharp_edge_line_art",
51
+ "duoline_style_icon",
52
+ "minimal_line_icon",
53
+ "stroke_only_glyph"
54
+ ],
55
+ "topics": [
56
+ "technology and engineering",
57
+ "scientific research",
58
+ "biomedical science",
59
+ "mathematics",
60
+ "business information",
61
+ "products and services",
62
+ "health facility",
63
+ "medical profession",
64
+ "scientific institution",
65
+ "government policy",
66
+ "architecture"
67
+ ]
68
+ },
69
+ "corporate_and_professional": {
70
+ "description": "商务与专业:传达信任、精确和可靠性,适用于商业报告、金融数据和技术文档。",
71
+ "keywords": [
72
+ "corporate_clean_icon",
73
+ "professional_minimal",
74
+ "business_infographic_style",
75
+ "data_viz_icon_style",
76
+ "tech_corporate_look",
77
+ "corporate_soft_gradient",
78
+ "formal_vector_shape",
79
+ "corporate_duotone",
80
+ "abstract_corporate_shape",
81
+ "clean_tech_icon",
82
+ "startup_style_icon",
83
+ "corporate_badge",
84
+ "enterprise_icon_style",
85
+ "presentation_icon_style"
86
+ ],
87
+ "topics": [
88
+ "business enterprise",
89
+ "business information",
90
+ "economy",
91
+ "market and exchange",
92
+ "products and services",
93
+ "employment",
94
+ "employment legislation",
95
+ "labour market",
96
+ "labour relations",
97
+ "retirement",
98
+ "government",
99
+ "government policy",
100
+ "law",
101
+ "health insurance",
102
+ "private health care",
103
+ "sport industry",
104
+ "sports management and ownership"
105
+ ]
106
+ },
107
+ "hand_drawn_and_organic": {
108
+ "description": "手绘与有机:具有人情味、创意和不完美感,适用于非正式、创意或个性化主题。",
109
+ "keywords": [
110
+ "hand_drawn_icon",
111
+ "sketchy_style_pictogram",
112
+ "doodle_icon_style",
113
+ "organic_shape_fill",
114
+ "imperfect_lines",
115
+ "chalkboard_style_icon",
116
+ "crayon_texture_fill",
117
+ "pencil_sketch_icon",
118
+ "hand_drawn_outline",
119
+ "crafty_paper_cut_style",
120
+ "whimsical_hand_drawn",
121
+ "marker_stroke_style",
122
+ "ink_sketch_glyph",
123
+ "rustic_hatch_fill",
124
+ "storybook_icon_style",
125
+ "organic_blob_shape",
126
+ "playful_hand_drawn"
127
+ ],
128
+ "topics": [
129
+ "arts and entertainment",
130
+ "culture",
131
+ "leisure",
132
+ "lifestyle",
133
+ "wellness",
134
+ "social learning",
135
+ "students",
136
+ "family",
137
+ "communities",
138
+ "social problem",
139
+ "non-governmental organisation (NGO)",
140
+ "conservation",
141
+ "parents group",
142
+ "values"
143
+ ]
144
+ },
145
+ "playful_and_cute": {
146
+ "description": "趣味与可爱:有趣、活泼、引人入胜,非常适合教育、社交媒体或轻松的主题。",
147
+ "keywords": [
148
+ "playful_rounded_shape",
149
+ "cute_cartoon_icon",
150
+ "kawaii_style_glyph",
151
+ "kid_friendly_vector",
152
+ "bouncy_form",
153
+ "fun_mascot_style",
154
+ "playful_doodle",
155
+ "toy_like_icon",
156
+ "playful_corporate_lite",
157
+ "educational_playful",
158
+ "soft_bubble_shape",
159
+ "childlike_drawing_style",
160
+ "sticker_style_icon",
161
+ "bubbly_pictogram",
162
+ "friendly_cartoon_icon",
163
+ "playful_badge",
164
+ "cute_minimal",
165
+ "rounded_corner_style"
166
+ ],
167
+ "topics": [
168
+ "social learning",
169
+ "school",
170
+ "students",
171
+ "parents group",
172
+ "family",
173
+ "leisure",
174
+ "lifestyle",
175
+ "birthday",
176
+ "anniversary",
177
+ "celebrity",
178
+ "arts and entertainment",
179
+ "products and services",
180
+ "religious festival and holiday",
181
+ "wellness"
182
+ ]
183
+ },
184
+ "tech_and_futuristic": {
185
+ "description": "科技与未来:现代、前卫、动感,暗示技术、数据、创新和科幻主题。",
186
+ "keywords": [
187
+ "tech_glow_outline",
188
+ "neon_line_art_icon",
189
+ "futuristic_hud_style",
190
+ "sci_fi_ui_glyph",
191
+ "data_stream_lines",
192
+ "circuit_board_pattern_fill",
193
+ "tech_minimal_glyph",
194
+ "digital_glitch_effect_icon",
195
+ "glossy_tech_icon",
196
+ "cyberpunk_style_pictogram",
197
+ "glowing_edge_effect",
198
+ "holographic_fill_style",
199
+ "tech_badge",
200
+ "dark_mode_ui_icon",
201
+ "gradient_line_art",
202
+ "plexus_style_lines",
203
+ "vector_circuit_icon",
204
+ "modern_tech_glyph",
205
+ "data_flow_abstract_shape",
206
+ "digital_network_icon"
207
+ ],
208
+ "topics": [
209
+ "technology and engineering",
210
+ "cyber warfare",
211
+ "scientific research",
212
+ "biomedical science",
213
+ "natural science",
214
+ "online and remote learning",
215
+ "business information",
216
+ "market and exchange",
217
+ "products and services",
218
+ "mass media",
219
+ "scientific institution"
220
+ ]
221
+ },
222
+ "retro_and_vintage": {
223
+ "description": "复古与怀旧:唤起过去的时代感,适用于历史数据、时间线或营造特定的怀旧氛围。",
224
+ "keywords": [
225
+ "retro_pixel_art_icon",
226
+ "8_bit_glyph_style",
227
+ "vintage_badge_design",
228
+ "retro_cartoon_style",
229
+ "mid_century_modern_icon",
230
+ "70s_groovy_style_vector",
231
+ "50s_atomic_era_shape",
232
+ "vintage_stamp_effect",
233
+ "retro_line_art_icon",
234
+ "woodcut_style_icon",
235
+ "engraving_hatch_fill",
236
+ "retro_tech_look",
237
+ "vintage_label_style",
238
+ "retro_halftone_fill",
239
+ "distressed_texture_overlay",
240
+ "retro_signage_style",
241
+ "vintage_script_accent"
242
+ ],
243
+ "topics": [
244
+ "culture",
245
+ "arts and entertainment",
246
+ "mass media",
247
+ "post-war reconstruction",
248
+ "social condition",
249
+ "leisure",
250
+ "lifestyle",
251
+ "products and services",
252
+ "history"
253
+ ]
254
+ },
255
+ "isometric_and_3d": {
256
+ "description": "等距与3D:增加深度和空间感,非常适合表现流程、地图、建筑或堆叠概念。",
257
+ "keywords": [
258
+ "isometric_vector_block",
259
+ "3d_icon_style",
260
+ "orthographic_view_icon",
261
+ "2.5d_style_pictogram",
262
+ "isometric_process_flow",
263
+ "soft_3d_clay_style",
264
+ "claymorphism_icon",
265
+ "glossy_3d_web_icon",
266
+ "vector_voxel_art",
267
+ "isometric_map_element",
268
+ "stacked_layers_style",
269
+ "3d_minimal_shape",
270
+ "neo_brutalism_3d_icon",
271
+ "low_poly_vector_icon",
272
+ "isometric_grid_style",
273
+ "3d_chart_icon",
274
+ "soft_ui_3d_style (neumorphism)",
275
+ "3d_cartoon_vector",
276
+ "vector_3d_render_style",
277
+ "soft_shadow_3d"
278
+ ],
279
+ "topics": [
280
+ "business enterprise",
281
+ "business information",
282
+ "technology and engineering",
283
+ "scientific research",
284
+ "emergency response",
285
+ "products and services",
286
+ "online and remote learning",
287
+ "market and exchange",
288
+ "logistics",
289
+ "supply chain",
290
+ "architecture",
291
+ "urban planning"
292
+ ]
293
+ },
294
+ "geometric_and_abstract": {
295
+ "description": "几何与抽象:概念性、模块化,使用基础形状(圆形、方形、三角形)构建图标。",
296
+ "keywords": [
297
+ "geometric_shape_build",
298
+ "abstract_pictogram",
299
+ "bauhaus_style_icon",
300
+ "modular_icon_design",
301
+ "low_poly_flat_icon",
302
+ "geometric_pattern_fill",
303
+ "sacred_geometry_lines",
304
+ "crystal_polygon_shape",
305
+ "memphis_style_elements",
306
+ "color_block_composition",
307
+ "tangram_style_icon",
308
+ "minimal_geometric_glyph",
309
+ "abstract_data_icon",
310
+ "geometric_logo_style",
311
+ "pattern_based_icon",
312
+ "triangular_mesh_icon",
313
+ "circles_and_squares_build",
314
+ "minimal_abstract_form",
315
+ "modernist_style_icon",
316
+ "geometric_badge_design"
317
+ ],
318
+ "topics": [
319
+ "mathematics",
320
+ "natural science",
321
+ "biomedical science",
322
+ "scientific research",
323
+ "scientific standards",
324
+ "technology and engineering",
325
+ "social sciences",
326
+ "demographics",
327
+ "economy",
328
+ "market and exchange",
329
+ "data visualization"
330
+ ]
331
+ },
332
+ "textured_and_grainy": {
333
+ "description": "纹理与颗粒:为扁平图标添加触感和深度,使用颗粒、纸张或喷漆等纹理。",
334
+ "keywords": [
335
+ "grain_texture_fill",
336
+ "noise_overlay_effect",
337
+ "spray_paint_texture",
338
+ "stipple_effect_fill",
339
+ "brushed_texture_icon",
340
+ "paper_texture_background",
341
+ "textured_shadow",
342
+ "craft_paper_style",
343
+ "flat_with_grain_shading",
344
+ "distressed_vector_icon",
345
+ "risograph_texture_effect",
346
+ "grunge_icon_style",
347
+ "canvas_texture_overlay",
348
+ "minimal_texture_accent",
349
+ "subtle_grain_overlay",
350
+ "sand_texture_fill",
351
+ "screen_print_effect_icon",
352
+ "sponge_paint_texture",
353
+ "chalky_texture"
354
+ ],
355
+ "topics": [
356
+ "arts and entertainment",
357
+ "culture",
358
+ "conservation",
359
+ "nature",
360
+ "lifestyle",
361
+ "leisure",
362
+ "products and services",
363
+ "food and drink",
364
+ "crafts",
365
+ "social condition"
366
+ ]
367
+ },
368
+ "eco_and_nature": {
369
+ "description": "生态与自然:强调有机、可持续和自然主题,常使用大地色系和流畅线条。",
370
+ "keywords": [
371
+ "organic_line_art_icon",
372
+ "botanical_icon_style",
373
+ "leaf_motif_icon",
374
+ "natural_form_shape",
375
+ "eco_friendly_badge",
376
+ "recycled_paper_texture_fill",
377
+ "hand_drawn_nature_icon",
378
+ "organic_flowing_lines",
379
+ "environmental_glyph",
380
+ "sustainable_icon",
381
+ "floral_line_art",
382
+ "woodgrain_pattern_fill",
383
+ "water_ripple_effect",
384
+ "soft_natural_shape",
385
+ "plant_silhouette_icon",
386
+ "outdoor_adventure_style",
387
+ "rustic_eco_icon"
388
+ ],
389
+ "topics": [
390
+ "climate change",
391
+ "conservation",
392
+ "environmental pollution",
393
+ "natural resource",
394
+ "nature",
395
+ "sustainability",
396
+ "wellness",
397
+ "lifestyle",
398
+ "government policy",
399
+ "leisure",
400
+ "welfare",
401
+ "agriculture"
402
+ ]
403
+ },
404
+ "health_and_wellness": {
405
+ "description": "健康与保健:干净、平静、柔和的风格,用于医疗、正念和生活方式等主题。",
406
+ "keywords": [
407
+ "clean_medical_icon",
408
+ "soft_rounded_shapes",
409
+ "calm_minimal_style",
410
+ "medical_glyph_style",
411
+ "heartbeat_line_art_icon",
412
+ "organic_wellness_shape",
413
+ "minimal_health_glyph",
414
+ "line_art_anatomy_icon",
415
+ "soft_gradient_fill_icon",
416
+ "pill_shape_design",
417
+ "yoga_pose_silhouette",
418
+ "mindfulness_icon_style",
419
+ "zen_style_brush_stroke",
420
+ "clinical_clean_line",
421
+ "medical_caduceus_style",
422
+ "spa_and_relaxation_icon",
423
+ "healthy_food_pictogram",
424
+ "scientific_clean_icon"
425
+ ],
426
+ "topics": [
427
+ "disease and condition",
428
+ "government health care",
429
+ "health facility",
430
+ "health insurance",
431
+ "health organisation",
432
+ "health treatment and procedure",
433
+ "medical profession",
434
+ "private health care",
435
+ "public health",
436
+ "wellness",
437
+ "lifestyle",
438
+ "biomedical science",
439
+ "social problem",
440
+ "family",
441
+ "leisure"
442
+ ]
443
+ },
444
+ "festive_and_celebratory": {
445
+ "description": "节日与庆典:明亮、欢快、有趣,用于假日、派对、公告和活动主题。",
446
+ "keywords": [
447
+ "confetti_pattern_fill",
448
+ "celebration_icon",
449
+ "party_doodle_style",
450
+ "holiday_glyph_set",
451
+ "sparkle_and_shine_accent",
452
+ "ribbon_and_banner_style",
453
+ "decorative_flourish_icon",
454
+ "bright_gradient_fill",
455
+ "carnival_style_pictogram",
456
+ "festive_badge_design",
457
+ "birthday_icon_style",
458
+ "event_pictogram_style",
459
+ "fireworks_burst_icon",
460
+ "playful_holiday_cartoon",
461
+ "invitation_style_glyph",
462
+ "gold_foil_accent",
463
+ "retro_party_style",
464
+ "seasonal_icon_pack",
465
+ "celebratory_burst_shape"
466
+ ],
467
+ "topics": [
468
+ "anniversary",
469
+ "award and prize",
470
+ "birthday",
471
+ "celebrity",
472
+ "ceremony",
473
+ "religious festival and holiday",
474
+ "sport event",
475
+ "sport achievement",
476
+ "leisure",
477
+ "lifestyle",
478
+ "family",
479
+ "communities",
480
+ "culture",
481
+ "record and achievement"
482
+ ]
483
+ },
484
+ "pop_art_and_comic": {
485
+ "description": "波普与漫画:大胆、醒目、高对比度,使用半色调圆点和粗黑轮廓等漫画技巧。",
486
+ "keywords": [
487
+ "pop_art_style_icon",
488
+ "comic_book_pictogram",
489
+ "halftone_dot_fill",
490
+ "bold_black_outline",
491
+ "dynamic_action_lines",
492
+ "ben_day_dots_fill",
493
+ "comic_speech_bubble_icon",
494
+ "pop_art_explosion_shape",
495
+ "graphic_high_contrast",
496
+ "retro_comic_style_icon",
497
+ "warhol_inspired_icon",
498
+ "screen_print_look_icon",
499
+ "comic_hatching_lines_fill",
500
+ "sticker_style_pop_art",
501
+ "bold_graphic_shape",
502
+ "pop_art_shadow_style",
503
+ "graphic_poster_style",
504
+ "cartoon_pop_art"
505
+ ],
506
+ "topics": [
507
+ "arts and entertainment",
508
+ "culture",
509
+ "mass media",
510
+ "celebrity",
511
+ "leisure",
512
+ "lifestyle",
513
+ "social problem",
514
+ "products and services",
515
+ "advertising",
516
+ "civil unrest",
517
+ "social condition"
518
+ ]
519
+ }
520
+ }
icon_generation/batch_icon_generator.py ADDED
@@ -0,0 +1,267 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ import json
3
+ from google import genai
4
+ from google.genai import types
5
+ from pathlib import Path
6
+ import time
7
+ from multiprocessing import Pool
8
+ from functools import partial
9
+
10
+ # API配置
11
+ API_KEY = os.getenv("GEMINI_API_KEY") or os.getenv("OPENAI_API_KEY", "")
12
+ client = genai.Client(
13
+ api_key=API_KEY,
14
+ http_options={"base_url": os.getenv("GEMINI_BASE_URL", "https://aihubmix.com/gemini")},
15
+ )
16
+
17
+ # 配置参数
18
+ ASPECT_RATIO = "3:2"
19
+ BATCH_SIZE = 24
20
+ MIN_BATCH_SIZE = 6
21
+ OUTPUT_DIR = "generated_icons"
22
+ NUM_PROCESSES = 3
23
+
24
+ def load_domain_attributes(file_path):
25
+ """读取domain_attributes.txt文件,返回扁平化的(domain, attribute)对列表"""
26
+ domain_attribute_pairs = []
27
+ current_domain = None
28
+
29
+ with open(file_path, 'r', encoding='utf-8') as f:
30
+ lines = f.readlines()
31
+
32
+ for line in lines:
33
+ line = line.strip()
34
+ if not line:
35
+ continue
36
+
37
+ # 检查是否是domain行(没有逗号的行)
38
+ if ',' not in line:
39
+ current_domain = line
40
+ else:
41
+ # 这是attributes行
42
+ if current_domain:
43
+ attributes = [attr.strip() for attr in line.split(',')]
44
+ for attr in attributes:
45
+ domain_attribute_pairs.append((current_domain, attr))
46
+
47
+ return domain_attribute_pairs
48
+
49
+ def load_templates(file_path):
50
+ """读取template_batch.json文件"""
51
+ with open(file_path, 'r', encoding='utf-8') as f:
52
+ templates = json.load(f)
53
+ return templates
54
+
55
+ def batch_domain_attribute_pairs(pairs, batch_size=BATCH_SIZE):
56
+ """将domain-attribute pairs分批,每批batch_size个"""
57
+ batches = []
58
+ for i in range(0, len(pairs), batch_size):
59
+ batch = pairs[i:i + batch_size]
60
+ batches.append(batch)
61
+ return batches
62
+
63
+ def generate_prompt(template, domain_attribute_pairs):
64
+ """生成prompt,将模板中的占位符替换为实际内容"""
65
+ # 格式化为: "domain1: attribute1, domain2: attribute2, ..."
66
+ pairs_text = ", ".join([f"{domain}: {attr}" for domain, attr in domain_attribute_pairs])
67
+ prompt = template.replace("{DOMAIN_ATTRIBUTE_PAIRS}", pairs_text)
68
+
69
+ # 添加固定的布局要求
70
+ actual_count = len(domain_attribute_pairs)
71
+ layout_requirement = (
72
+ f" The output must be a single image with an exact 6:4 aspect ratio (landscape orientation, width greater than height). "
73
+ f"The image must contain exactly {actual_count} icons arranged in a strict grid of 6 columns (horizontal, left to right) and 4 rows (vertical, top to bottom). "
74
+ f"Do not rotate, transpose, or alter the grid orientation. "
75
+ f"No text, letters, numbers, labels, captions, or icon titles. Each icon must not include any titles or written elements. "
76
+ f"Use a pure white background only."
77
+ )
78
+
79
+ prompt = prompt + layout_requirement
80
+ return prompt
81
+
82
+ def generate_icons_for_batch(batch_idx, domain_attribute_pairs, templates, output_dir):
83
+ """为单个batch生成所有风格的icons"""
84
+ batch_dir = os.path.join(output_dir, f"batch_{batch_idx:04d}")
85
+ os.makedirs(batch_dir, exist_ok=True)
86
+
87
+ print(f"\n{'='*80}")
88
+ print(f"批次 {batch_idx} (含 {len(domain_attribute_pairs)} 个 domain-attribute pairs)")
89
+
90
+ success_count = 0
91
+ failed_count = 0
92
+ skipped_count = 0
93
+
94
+ # 为每个style生成icons
95
+ for style_name, template in templates.items():
96
+ print(f"\n 风格: {style_name}")
97
+
98
+ # 生成文件名
99
+ image_filename = f"batch_{batch_idx:04d}_{style_name}.png"
100
+ annotation_filename = f"batch_{batch_idx:04d}_{style_name}.txt"
101
+
102
+ image_path = os.path.join(batch_dir, image_filename)
103
+ annotation_path = os.path.join(batch_dir, annotation_filename)
104
+
105
+ # 断点续传:检查文件是否已存在
106
+ if os.path.exists(image_path) and os.path.exists(annotation_path):
107
+ print(f" ⏭️ 跳过(文件已存在): {image_filename}")
108
+ skipped_count += 1
109
+ success_count += 1 # 已存在的文件计入成功数
110
+ continue
111
+
112
+ try:
113
+ # 生成prompt
114
+ prompt = generate_prompt(template, domain_attribute_pairs)
115
+
116
+ # 每个进程需要创建自己的API客户端
117
+ client = genai.Client(
118
+ api_key=API_KEY,
119
+ http_options={"base_url": "https://aihubmix.com/gemini"},
120
+ )
121
+
122
+ # 调用API生成图像
123
+ response = client.models.generate_content(
124
+ model="gemini-3-pro-image-preview",
125
+ contents=prompt,
126
+ config=types.GenerateContentConfig(
127
+ response_modalities=['TEXT', 'IMAGE'],
128
+ image_config=types.ImageConfig(
129
+ aspect_ratio=ASPECT_RATIO
130
+ ),
131
+ ),
132
+ )
133
+
134
+ # 保存图像和文本
135
+ for part in response.parts:
136
+ if part.text:
137
+ print(f" 生成说明: {part.text[:100]}...")
138
+ elif image := part.as_image():
139
+ image.save(image_path)
140
+ print(f" ✅ 图像已保存: {image_filename}")
141
+
142
+ # 保存标注
143
+ save_annotation(annotation_path, domain_attribute_pairs, style_name)
144
+ print(f" ✅ 标注已保存: {annotation_filename}")
145
+
146
+ success_count += 1
147
+
148
+ except Exception as e:
149
+ print(f" ❌ 生成失败: {str(e)}")
150
+ failed_count += 1
151
+ continue
152
+
153
+ if skipped_count > 0:
154
+ print(f"\n 批次 {batch_idx} 完成: ✅ {success_count} 成功 (含 {skipped_count} 个跳过), ❌ {failed_count} 失败")
155
+ else:
156
+ print(f"\n 批次 {batch_idx} 完成: ✅ {success_count} 成功, ❌ {failed_count} 失败")
157
+ return success_count, failed_count
158
+
159
+ def save_annotation(file_path, domain_attribute_pairs, style):
160
+ """保存txt标注文件,格式:第一行style,后续每行domain, attribute"""
161
+ with open(file_path, 'w', encoding='utf-8') as f:
162
+ f.write(f"Style: {style}\n")
163
+ for domain, attr in domain_attribute_pairs:
164
+ f.write(f"{domain}, {attr}\n")
165
+
166
+ def process_single_batch(args):
167
+ """处理单个batch的所有风格(用于并发处理)"""
168
+ batch_idx, batch, templates, output_dir = args
169
+
170
+ # 跳过少于MIN_BATCH_SIZE的最后一批(如果不是第一批)
171
+ if len(batch) < MIN_BATCH_SIZE and batch_idx > 1:
172
+ print(f"\n⚠️ 跳过批次 {batch_idx}: 只有 {len(batch)} 个pairs (少于最小值 {MIN_BATCH_SIZE})")
173
+ return 0, 0, True # success_count, failed_count, skipped
174
+
175
+ success, failed = generate_icons_for_batch(batch_idx, batch, templates, output_dir)
176
+ return success, failed, False
177
+
178
+ def main():
179
+ """主函数"""
180
+ print("="*80)
181
+ print("批量图标生成器 (扁平化模式)")
182
+ print("="*80)
183
+
184
+ # 检查文件是否存在
185
+ domain_file = "domain_attributes.txt"
186
+ template_file = "template_batch.json"
187
+
188
+ if not os.path.exists(domain_file):
189
+ print(f"❌ 错误: 找不到文件 {domain_file}")
190
+ return
191
+
192
+ if not os.path.exists(template_file):
193
+ print(f"❌ 错误: 找不到文件 {template_file}")
194
+ return
195
+
196
+ # 创建输出目录
197
+ os.makedirs(OUTPUT_DIR, exist_ok=True)
198
+
199
+ # 加载数据
200
+ print(f"\n📖 加载domain-attribute pairs...")
201
+ domain_attribute_pairs = load_domain_attributes(domain_file)
202
+ print(f" 共加载 {len(domain_attribute_pairs)} 个 domain-attribute pairs")
203
+
204
+ print(f"\n📖 加载模板...")
205
+ templates = load_templates(template_file)
206
+ print(f" 共加载 {len(templates)} 个风格模板: {', '.join(templates.keys())}")
207
+
208
+ # 配置信息
209
+ print(f"\n⚙️ 配置:")
210
+ print(f" 长宽比: {ASPECT_RATIO}")
211
+ print(f" 单批最大数量: {BATCH_SIZE}")
212
+ print(f" 单批最小数量: {MIN_BATCH_SIZE}")
213
+ print(f" 并发进程数: {NUM_PROCESSES}")
214
+ print(f" 输出目录: {OUTPUT_DIR}")
215
+
216
+ # 分批
217
+ batches = batch_domain_attribute_pairs(domain_attribute_pairs, BATCH_SIZE)
218
+ print(f"\n📦 分成 {len(batches)} 个批次")
219
+
220
+ # 扫描已存在的文件(断点续传预检)
221
+ print(f"\n🔍 扫描已存在的文件...")
222
+ existing_files = 0
223
+ total_expected_files = 0
224
+ for batch_idx, batch in enumerate(batches, 1):
225
+ if len(batch) < MIN_BATCH_SIZE and batch_idx > 1:
226
+ continue
227
+ batch_dir = os.path.join(OUTPUT_DIR, f"batch_{batch_idx:04d}")
228
+ for style_name in templates.keys():
229
+ total_expected_files += 1
230
+ image_filename = f"batch_{batch_idx:04d}_{style_name}.png"
231
+ annotation_filename = f"batch_{batch_idx:04d}_{style_name}.txt"
232
+ image_path = os.path.join(batch_dir, image_filename)
233
+ annotation_path = os.path.join(batch_dir, annotation_filename)
234
+ if os.path.exists(image_path) and os.path.exists(annotation_path):
235
+ existing_files += 1
236
+
237
+ print(f" 已存在: {existing_files}/{total_expected_files} 个文件")
238
+ print(f" 需要生成: {total_expected_files - existing_files} 个文件")
239
+
240
+ # 准备并发任务
241
+ tasks = []
242
+ for batch_idx, batch in enumerate(batches, 1):
243
+ tasks.append((batch_idx, batch, templates, OUTPUT_DIR))
244
+
245
+ # 使用进程池并发处理多个batch
246
+ print(f"\n🚀 使用 {NUM_PROCESSES} 个进程并发处理批次...")
247
+ print(f"💡 提示: 已存在的文件将自动跳过(断点续传)")
248
+ total_success = 0
249
+ total_failed = 0
250
+
251
+ with Pool(processes=NUM_PROCESSES) as pool:
252
+ results = pool.map(process_single_batch, tasks)
253
+
254
+ # 统计结果
255
+ for success, failed, skipped in results:
256
+ if not skipped:
257
+ total_success += success
258
+ total_failed += failed
259
+
260
+ print(f"\n{'='*80}")
261
+ print("✅ 全部生成完成!")
262
+ print(f"总计: ✅ {total_success} 成功, ❌ {total_failed} 失败")
263
+ print(f"输出目录: {OUTPUT_DIR}")
264
+ print("="*80)
265
+
266
+ if __name__ == "__main__":
267
+ main()
icon_generation/count.py ADDED
@@ -0,0 +1,52 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """
3
+ 统计 generated_icons 目录下各个子目录中的 PNG 文件数量
4
+ """
5
+
6
+ import os
7
+ from pathlib import Path
8
+
9
+ def count_png_files():
10
+ # 设置 generated_icons 目录路径
11
+ base_dir = Path(__file__).parent / "generated_icons"
12
+
13
+ if not base_dir.exists():
14
+ print(f"错误: 目录 {base_dir} 不存在")
15
+ return
16
+
17
+ # 存储每个子目录的统计结果
18
+ subdirs_stats = {}
19
+ total_png_count = 0
20
+
21
+ # 遍历所有子目录
22
+ for subdir in sorted(base_dir.iterdir()):
23
+ if subdir.is_dir():
24
+ # 统计当前子目录下的 PNG 文件数量
25
+ png_files = list(subdir.glob("*.png"))
26
+ png_count = len(png_files)
27
+
28
+ subdirs_stats[subdir.name] = png_count
29
+ total_png_count += png_count
30
+
31
+ # 打印结果
32
+ print("=" * 50)
33
+ print(f"{'子目录':<20} {'PNG 文件数量':>15}")
34
+ print("=" * 50)
35
+
36
+ for subdir_name, count in subdirs_stats.items():
37
+ print(f"{subdir_name:<20} {count:>15}")
38
+
39
+ print("=" * 50)
40
+ print(f"{'总计':<20} {total_png_count:>15}")
41
+ print("=" * 50)
42
+
43
+ # 额外统计信息
44
+ print(f"\n子目录总数: {len(subdirs_stats)}")
45
+ print(f"PNG 文件总数: {total_png_count}")
46
+
47
+ if subdirs_stats:
48
+ avg_png_per_dir = total_png_count / len(subdirs_stats)
49
+ print(f"平均每个子目录的 PNG 数量: {avg_png_per_dir:.2f}")
50
+
51
+ if __name__ == "__main__":
52
+ count_png_files()
icon_generation/domain_attributes.txt ADDED
@@ -0,0 +1,957 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Football Club
2
+ Real Madrid, FC Barcelona, Manchester United, Manchester City, Liverpool FC, Chelsea FC, Arsenal FC, Bayern Munich, Borussia Dortmund, Paris Saint-Germain, Juventus, AC Milan, Inter Milan, AS Roma, Napoli, Atletico Madrid, Ajax, PSV Eindhoven, Benfica, FC Porto, Sporting CP, Celtic FC, Rangers FC, Galatasaray, Fenerbahce, Boca Juniors, River Plate, Flamengo, Santos FC
3
+
4
+ Basketball Club
5
+ Los Angeles Lakers, Boston Celtics, Golden State Warriors, Chicago Bulls, Miami Heat, San Antonio Spurs, Brooklyn Nets, New York Knicks, Philadelphia 76ers, Toronto Raptors, Dallas Mavericks, Houston Rockets, Phoenix Suns, Denver Nuggets, Milwaukee Bucks, Cleveland Cavaliers, Detroit Pistons, Atlanta Hawks, Indiana Pacers, Orlando Magic, Utah Jazz, Oklahoma City Thunder, Portland Trail Blazers, Sacramento Kings, LA Clippers, Minnesota Timberwolves, Memphis Grizzlies, New Orleans Pelicans, Washington Wizards
6
+
7
+ Education Level
8
+ Primary Education, Secondary Education, Upper Secondary Education, Vocational Education, Technical Education, Associate Degree, Bachelor’s Degree, Master’s Degree, Doctoral Degree, Postdoctoral Education, Professional Degree, Teacher Training, Special Education, Adult Education, Continuing Education, Distance Learning, Online Education, Homeschooling, Early Childhood Education, Preschool, Kindergarten
9
+
10
+ Currency
11
+ US Dollar, Euro, British Pound, Japanese Yen, Chinese Yuan, Canadian Dollar, Australian Dollar, Swiss Franc, Indian Rupee, Russian Ruble, South Korean Won, Singapore Dollar, Hong Kong Dollar, New Zealand Dollar, Brazilian Real, Mexican Peso, Argentine Peso, Chilean Peso, Colombian Peso, South African Rand, Turkish Lira, Israeli Shekel, Saudi Riyal, UAE Dirham, Qatari Riyal, Kuwaiti Dinar, Bahraini Dinar, Omani Rial, Egyptian Pound, Nigerian Naira
12
+
13
+ Landmark
14
+ Eiffel Tower, Statue of Liberty, Great Wall of China, Taj Mahal, Colosseum, Machu Picchu, Petra, Christ the Redeemer, Big Ben, Sydney Opera House, Burj Khalifa, Golden Gate Bridge, Mount Fuji, Angkor Wat, Acropolis, Sagrada Familia, Alhambra, Stonehenge, Moai Statues, Chichen Itza, Mount Rushmore, Notre-Dame Cathedral, Louvre Museum, Buckingham Palace, Tower of London, Brandenburg Gate, Neuschwanstein Castle, Blue Mosque, Hagia Sophia, Grand Canyon
15
+
16
+ Esports Game
17
+ League of Legends, Dota 2, Counter-Strike 2, Valorant, Overwatch 2, Fortnite, PUBG: Battlegrounds, Apex Legends, Call of Duty, Rainbow Six Siege, StarCraft II, Warcraft III, Hearthstone, Rocket League, FIFA, EA Sports FC, Street Fighter, Tekken, Super Smash Bros., Smash Ultimate, Mobile Legends: Bang Bang, Arena of Valor, Free Fire, Clash Royale, Brawl Stars, Pokémon Unite, Teamfight Tactics, Age of Empires II, Halo Infinite
18
+
19
+ Trend Direction
20
+ Upward Trend, Downward Trend, Stable Trend, Volatile Trend, Fluctuating Trend, Seasonal Increase, Seasonal Decrease, Cyclical Trend, Exponential Growth, Linear Growth, Rapid Growth, Slow Growth, Plateau, Peak, Decline, Recovery, Rebound, Correction, Consolidation, Breakout
21
+
22
+ Transportation Mode
23
+ Car, Bus, Train (Rail), Subway (Metro), Bicycle, Walking, Taxi / Rideshare, Airplane (Aircraft), Ferry, Truck
24
+
25
+ Smartphone Brand
26
+ Apple, Samsung, Xiaomi, Oppo, Vivo, Google, Motorola, Huawei, OnePlus, Nokia
27
+
28
+ Vehicle Type
29
+ SUV, Sedan, Hatchback, Pickup Truck, Van, Bus, Motorcycle, Bicycle, Coupe, Semi-trailer Truck
30
+
31
+ Energy Technologies
32
+ Solar PV, Onshore wind, Hydropower, Nuclear power, Natural gas power, Lithium-ion battery, Hydrogen (production & storage), Grid transmission & distribution, Energy efficiency (LED lighting), Electric vehicles
33
+
34
+ Energy Source
35
+ Oil, Natural Gas, Coal, Hydropower, Solar, Wind, Nuclear, Biomass, Biofuel, Hydrogen, Geothermal, Biogas, Municipal Solid Waste, Tidal, Wave, Peat, Wood, Synthetic Fuels, Ocean Thermal Energy Conversion, Other Renewables
36
+
37
+ Geographic Region
38
+ Urban, Rural, Suburban, Coastal, Mountain, Plains, Forest, Desert, Island, Wetland
39
+
40
+ Academic Subject
41
+ Mathematics, Science, English, History, Biology, Chemistry, Physics, Computer Science, Economics, Psychology, Art, Music, Physical Education, Foreign Languages, Geography, Business, Engineering, Education, Philosophy, Political Science, Sociology, Statistics, Data Science, Information Technology, Environmental Science, Health Sciences, Finance, Accounting, Law, Architecture, Anthropology, Marketing, Media Studies, Religious Studies, Agriculture, Humanities
42
+
43
+ Material
44
+ Plastic, Wood, Steel, Concrete, Glass, Paper, Aluminum, Cotton, Rubber, Ceramic
45
+
46
+ Product Brand
47
+ Apple, Amazon, Google, Microsoft, Samsung, Nike, Coca-Cola, Toyota, Sony, McDonald's, Mercedes-Benz, BMW, Adidas, Ford, Honda, Louis Vuitton, Gucci, H&M, Zara, Uniqlo, Rolex, Tesla, LG, Intel, Dell, HP, Nintendo, Spotify, Skechers, New Balance, Puma, Converse, Reebok, ASICS, Under Armour, Tommy Hilfiger, Razer, Logitech, Corsair, SteelSeries, HyperX, Turtle Beach, Astro, Callaway, Brooks, Saucony, Carhartt, Diesel, Maserati, Fendi, Salomon, G-Star RAW, American Eagle
48
+
49
+ Client Engagement Channel
50
+ Website / Online Portal, Mobile App, Email, SMS / Text Messaging, Phone / Call Center, Live Chat / Chatbot, Branch / In-person Banking, Social Media, Push Notification, ATM / Kiosk, Financial Advisor, Video Conference, Mobile Web, Brokerage Platform, Robo-advisor / Digital Advisor, Partner / Channel Partner, API / Developer Integration, Direct Mail / Postal Mail, In-person Events / Seminars, Voice Assistant
51
+
52
+ Park Type
53
+ Urban Park, Playground, National Park, State Park, Botanical Garden, Zoo, Aquarium, Amusement Park, Nature Reserve, Campground
54
+
55
+ Camera Form Factor
56
+ Smartphone Camera, Mirrorless Camera, DSLR, Point-and-Shoot, Action Camera, Camcorder, Instant Camera, Film Camera, 360 Camera, Underwater/Waterproof Camera
57
+
58
+ Residential Room Type
59
+ Kitchen, Bathroom, Bedroom, Living Room, Dining Room, Garage, Laundry Room, Basement, Home Office, Closet
60
+
61
+ Business Segment (Food & Beverage)
62
+ Beverages, Snacks, Dairy, Bakery, Fresh Produce, Frozen Foods, Meat & Poultry, Prepared Foods, Confectionery, Refrigerated Foods, Plant-Based Foods, Food Ingredients & Commodities, Condiments & Sauces, Cereals & Grains, Seafood, Baby & Infant Nutrition, Pet Food, Alcoholic Beverages, Health & Functional Foods, Foodservice & Catering, Oils & Fats, Packaged & Shelf-stable Foods, Sugar & Sweeteners, Spices & Seasonings, Organic Foods, Ethnic & Specialty Foods, Retail Grocery, Logistics & Distribution, Packaging Materials & Solutions, Nutritionals & Supplements
63
+
64
+ Leisure Activity
65
+ Watching TV and Movies, Listening to Music, Socializing, Reading, Cooking, Fitness / Exercise, Video Gaming, Traveling, Gardening, Hiking
66
+
67
+ Museum Object Type
68
+ Ceramics, Textiles, Paintings, Sculptures, Prints and Drawings, Photographs, Furniture, Jewelry, Metalwork, Manuscripts, Books, Coins, Tools, Archaeological Materials, Weapons, Glass, Musical Instruments, Costume, Architectural Elements, Maps and Charts, Scientific Instruments, Ceremonial Objects, Inscribed Tablets, Ephemera, Time-based Media, Models and Replicas, Toys and Games, Medals and Tokens
69
+
70
+ Pollution Source or Pollutant Type
71
+ Particulate matter (PM2.5/PM10), Nitrogen oxides (NOx), Sulfur dioxide, Carbon monoxide, Ozone, Volatile organic compounds (VOCs), Plastics / microplastics, Agricultural runoff, Sewage / wastewater discharge, Industrial effluent (chemical waste), Pesticides, Nitrates, Phosphates, Urban runoff / stormwater, Oil spills, Heavy metals, Greenhouse gases (CO2, CH4), Black carbon (soot), Sediment (erosion/siltation), Toxic chemical spills, Radioactive contamination, Thermal pollution, Ammonia, Light pollution, Noise pollution
72
+
73
+ Medical Injury Type
74
+ Fracture, Sprain, Strain, Laceration, Burn, Concussion, Dislocation, Spinal cord injury, Traumatic brain injury, Amputation
75
+
76
+ Dietary Preference or Restriction
77
+ No Dietary Restrictions, Vegetarian, Gluten-Free, Dairy-Free, Halal, Vegan, Lactose-Free, Nut-Free, Kosher, Organic, Peanut-Free, Low-Carb, Keto, Low-Sodium, Pescatarian, Diabetic-Friendly, Soy-Free, Egg-Free, Shellfish-Free, Fish-Free, Sugar-Free, Paleo, Low-FODMAP
78
+
79
+ Dietary Protein Source
80
+ Chicken, Beef, Pork, Fish, Eggs, Milk, Beans, Tofu, Nuts, Shrimp
81
+
82
+ Lighting Type
83
+ LED, Fluorescent, Incandescent, Halogen, High-Intensity Discharge, Neon, Solar-powered, Smart Lighting, Fiber Optic
84
+
85
+ Seafood Species
86
+ Shrimp, Salmon, Tuna, Cod, Crab, Lobster, Oyster, Squid, Scallop, Mackerel
87
+
88
+ Travel Category
89
+ Flights, Hotels, Vacation Rentals, Car Rentals, Trains, Buses, Cruises, Tours & Activities, Travel Insurance, Travel Packages
90
+
91
+ Cryptocurrency Name
92
+ Bitcoin, Ethereum, Tether, USD Coin, XRP, Solana, Cardano, Dogecoin, Litecoin, Binance Coin
93
+
94
+ Food Category
95
+ Vegetables, Fruits, Grains, Dairy, Meat, Beverages, Snacks, Baked Goods, Seafood, Poultry, Eggs, Frozen Foods, Prepared Meals, Condiments & Sauces, Oils & Fats, Spices & Herbs, Canned & Preserved Foods, Sweets & Confectionery, Desserts, Legumes & Pulses, Nuts & Seeds, Breakfast Foods, Soups & Stews, Baby Food, Alcoholic Beverages, Plant-based Alternatives, Fast Food, Fermented Foods, Salads & Ready-to-eat Bowls, Cereals & Breakfast Grains
96
+
97
+ Fashion Accessory Category
98
+ Jewelry, Handbags, Wallets, Sunglasses, Watches, Hats, Scarves, Belts, Gloves, Hair Accessories
99
+
100
+ Tourism Type
101
+ Leisure tourism, Domestic tourism, International tourism, Business tourism, Cultural tourism, Adventure tourism, Ecotourism, Wellness tourism, Religious/pilgrimage tourism, Cruise tourism
102
+
103
+ Interest Category
104
+ Sports, Music, Movies, Television, Travel, Food & Cooking, Social Media, Fitness & Wellness, Gaming, Technology, Books & Reading, Fashion, Photography, Art, Arts & Crafts, Programming, Cars & Automotive, Science, History, Finance & Investing, Business & Entrepreneurship, Gardening, Pets & Animals, DIY & Home Improvement, Outdoor Activities, Board Games, Theater & Performing Arts, Dance, Languages, Comics & Graphic Novels, Collecting, Yoga & Meditation, Parenting & Family, Politics & Current Affairs, Tabletop Role-Playing Games, Education & Learning
105
+
106
+ Financial Incentive Type
107
+ Tax Credit, Tax Deduction, Rebate, Discount, Cashback, Grant, Subsidy, Voucher, Low-Interest Loan, Loan Guarantee
108
+
109
+ Museum Focus
110
+ Art, History, Natural History, Science, Children's Museum, Archaeology, Anthropology & Ethnography, Decorative Arts, Design, Contemporary Art, Photography, Technology & Industry, Maritime, Military, Historic House, Heritage, Religious, Music, Sports, Transportation, Aviation, Numismatics, Textiles, Fashion, Medical & Health, Space & Astronomy, Open-air / Living History, Specialty / Single-subject
111
+
112
+ Waste Material Type
113
+ Municipal solid waste, Food waste, Plastic, Paper, Yard waste, Cardboard, Construction and demolition waste, Glass, Metal, Industrial waste, Concrete, Electronic waste, Textiles, Used oil, Batteries, Sewage sludge, Hazardous waste, Paints and solvents, Rubber, Wood, Tires, Medical waste, Pharmaceutical waste
114
+
115
+ Freight Cargo Type
116
+ Containers, Crude oil, Refined petroleum products, Coal, Iron ore, Grain, Chemicals (liquid bulk), Automobiles (Ro-Ro), Refrigerated goods (Reefer cargo), Project cargo / Heavy lift / Oversize
117
+
118
+ Funding Source
119
+ Donations, Grants, Sponsorships, Ticket Sales, Membership Fees, Government Contracts, Loans, Product Sales, Service Fees, Investment Income
120
+
121
+ Civil Engineering Structure Type
122
+ Bridge, Tunnel, Road (Overpass/Underpass/Viaduct), Dam, Levee, Seawall, Canal, Retaining Wall, Pier/Wharf/Jetty, Reservoir
123
+
124
+ Recipe Ingredient
125
+ Salt, Olive Oil, Onion, Garlic, Eggs, Flour, Sugar, Milk, Butter, Chicken
126
+
127
+ Software Application Name
128
+ Google Chrome, Gmail, YouTube, WhatsApp, Facebook, Instagram, Microsoft Word, Microsoft Excel, Zoom, Google Drive
129
+
130
+ Sofa Type
131
+ Sectional, Sleeper sofa, Loveseat, Reclining sofa, Modular sofa, Futon, Chaise lounge, Chesterfield, Three-seater sofa, Settee
132
+
133
+ Sales Offer Type
134
+ Discount, Coupon / Promo Code, Free Shipping, Buy One Get One (BOGO), Bundle, Clearance Sale, Flash Sale, Free Trial, Cashback, Gift with Purchase
135
+
136
+ Financial Product Type
137
+ Checking Account, Savings Account, Credit Card, Debit Card, Mortgage, Personal Loan, Auto Loan, Brokerage Account, Certificate of Deposit (CD), Individual Retirement Account (IRA)
138
+
139
+ Cuisine
140
+ Italian, Chinese, Mexican, American, Indian, Japanese, Thai, Mediterranean, French, Middle Eastern
141
+
142
+ Season (Calendar & Climatic)
143
+ Spring, Summer, Autumn, Winter, Holiday season, Rainy/Wet season, Dry season, Hurricane/Typhoon/Cyclone season, Flu season, Tourist season
144
+
145
+ Company Name
146
+ Apple, Amazon, Google, Microsoft, Meta, Walmart, Tesla, Samsung, Toyota, Coca-Cola
147
+
148
+ Live Performance Type
149
+ Concert, Play, Musical, Stand-up Comedy, Dance, Ballet, Opera, Children's Theatre, Puppetry, Magic Show, Circus, Cabaret, Performance Art, Variety Show, Spoken Word / Poetry Slam, Improv Comedy, Physical Theatre, Orchestra / Symphony, Choral Performance, Recital, One-person Show (Monologue), Street Performance / Busking, Drag Show, Burlesque, Pantomime, Mime, Immersive / Interactive Theatre, Site-specific Performance, Multimedia / Digital Performance, Educational / School Performance
150
+
151
+ Animal Species
152
+ Dog, Cat, Chicken, Cattle, Horse, Pig, Elephant, Lion, Tiger, Bear
153
+
154
+ Event Type
155
+ Meeting, Conference, Workshop, Seminar, Concert, Festival, Sporting Event, Exhibition, Market/Fair, Parade, Protest/Rally, Wedding, Theater Performance, Film Screening, Dance Performance, Class/Training, Webinar/Virtual Event, Building Fire, Wildfire, Flood, Thunderstorm, Hailstorm, Heatwave, Drought, Tropical Cyclone, Tornado, Earthquake, Tsunami, Landslide, Sandstorm, Severe Winter Storm, Volcanic Eruption, Chemical Spill, Industrial Accident, Power Outage, Traffic Accident, Explosion, Mass Casualty Incident, Evacuation
156
+
157
+ Mobile Phone Type
158
+ Smartphone, Feature phone, Landline phone, VoIP phone, Flip phone, Rugged phone, Satellite phone, Desk phone
159
+
160
+ Accommodation Type
161
+ House, Apartment, Hotel, Vacation Rental, Condominium, Hostel, Bed and Breakfast, Resort, Motel, Cabin
162
+
163
+ Operating System
164
+ Android, iOS, Windows, macOS, Linux, Chrome OS
165
+
166
+ Vehicle Powertrain Type
167
+ Gasoline (ICE), Diesel (ICE), Battery Electric (BEV), Hybrid (HEV), Plug-in Hybrid (PHEV), Fuel Cell (FCEV), Natural Gas (CNG), Flex-Fuel (E85)
168
+
169
+ Industry Sector
170
+ Healthcare, Manufacturing, Technology, Finance, Retail, Construction, Education, Hospitality & Tourism, Real Estate, Transportation & Logistics, Energy, Agriculture, Automotive, Pharmaceuticals & Biotechnology, Food & Beverage, Consumer Goods (FMCG), Telecommunications, Media & Entertainment, Professional & Business Services, Government / Public Sector, Mining & Oil & Gas, Chemicals, Aerospace & Defense, Electronics & Electrical Equipment, Industrial Machinery & Equipment, Renewable Energy, Utilities, Software & SaaS, Consumer Electronics, Textiles & Apparel, Nonprofit / Social Services, Advertising & Marketing Services
171
+
172
+ Web Browser
173
+ Chrome, Safari, Edge, Firefox, Opera, Samsung Internet, UC Browser, Internet Explorer, Brave, Tor Browser
174
+
175
+ Entertainment Genre
176
+ Drama, Comedy, Action, Thriller, Crime, Romance, Romantic Comedy, Mystery, Horror, Documentary, Animation, Family, Adventure, Fantasy, Science Fiction, Superhero, Biography, History, Musical, War, Western, Reality, Sports, Coming-of-Age, Noir, Satire, Art House, Experimental
177
+
178
+ Media Format
179
+ Digital, Print, Video, Audio, Streaming, Television, Film, Podcast, Video Game, Live Performance
180
+
181
+ Smart Home Device Category
182
+ Smart Speakers & Voice Assistants, Smart Lighting, Smart Thermostats & Climate Control, Security Cameras, Video Doorbells, Smart Locks, Smart Plugs & Outlets, Smart Appliances, Smart Sensors (motion, contact, leak), Home Automation Hubs & Controllers
183
+
184
+ Art Medium
185
+ Painting, Drawing, Photography, Sculpture, Digital Art, Printmaking, Mixed Media, Ceramics, Textiles (Fiber Art), Watercolor, Oil Painting, Acrylic Painting, Pastel, Charcoal, Ink, Collage, Illustration, Glass, Metalwork, Woodwork, Encaustic, Fresco, Mosaic, Tapestry, Video Art, Film, Performance Art, Installation, Screenprint
186
+
187
+ Product Category
188
+ Electronics, Apparel, Food & Grocery, Home & Kitchen, Health & Personal Care, Books, Furniture, Toys & Games, Sports & Outdoors, Automotive
189
+
190
+ Land Use Type
191
+ Agriculture, Forest, Grassland, Wetland, Water, Residential, Commercial, Industrial, Transportation, Mining / Quarry
192
+
193
+ Device Type
194
+ Smartphone, Laptop, Desktop Computer, Tablet, Smart TV, Smartwatch, Gaming Console, Headphones, Printer, Router
195
+
196
+ Crop Type
197
+ Corn, Wheat, Rice, Soybean, Potato, Sugarcane, Cotton, Barley, Canola/Rapeseed, Peanut
198
+
199
+ Coffee Brewing Method
200
+ Drip Coffee, Espresso, Pour-over, French Press, Cold Brew, Instant Coffee, Pod Coffee, Moka Pot, Turkish Coffee
201
+
202
+ Social Event Type
203
+ Birthday Party, Wedding, Holiday Party, Dinner Party, Graduation Party, Baby Shower, Funeral / Memorial Service, House Party, Retirement Party
204
+
205
+ Purchase Channel
206
+ E-commerce Website, In-store (Retail), Mobile App, Third-party Marketplace, Click & Collect (Buy Online, Pick Up In Store), Subscription / Recurring Order, Telephone (Call Center / Phone Order), Wholesale / Distributor (B2B Channel), Social Commerce (Social Media Shops), Self-service Kiosk (In-store Kiosk)
207
+
208
+ Investment Asset Class
209
+ Equities, Fixed Income, Cash and Cash Equivalents, Real Estate, Commodities, Currencies (Foreign Exchange), Derivatives, Private Equity, Hedge Funds, Cryptocurrencies
210
+
211
+ Expense Category
212
+ Groceries, Rent, Utilities, Transportation, Restaurants & Dining, Fuel, Insurance, Healthcare & Medical, Taxes, Subscriptions & Streaming, Entertainment, Clothing & Apparel, Household Goods & Supplies, Home Maintenance & Repairs, Travel & Accommodation, Public Transport, Taxi & Ride-sharing, Parking & Tolls, Childcare & Education, Personal Care & Grooming, Gifts & Donations, Coffee & Snacks, Alcohol & Bars, Pet Care & Supplies, Professional Services, Office Supplies, Events & Catering, Venue Hire, Photography & Videography, Flowers & Decorations
213
+
214
+ Music Genre
215
+ Pop, Rock, Hip Hop, R&B, Electronic, Country, Latin, Classical, Jazz, Reggae
216
+
217
+ Dwelling Type
218
+ Single-Family Home, Apartment, Condominium, Townhouse, Multi-Family Home, Duplex, Manufactured Home, Studio Apartment, Cabin, Tiny House
219
+
220
+ Time of Day
221
+ Morning, Afternoon, Evening, Night, Dawn, Noon, Dusk, Midnight
222
+
223
+ Produce (Fruit & Vegetable)
224
+ Tomato, Potato, Onion, Banana, Apple, Lettuce, Carrot, Corn, Garlic, Cucumber, Bell pepper, Broccoli, Spinach, Avocado, Grapes, Orange, Strawberry, Mango, Lemon, Lime, Sweet potato, Cabbage, Cauliflower, Zucchini, Eggplant, Chili pepper, Pea, Green bean, Mushroom, Kale, Pumpkin, Pear, Peach, Pineapple, Watermelon, Melon, Beet, Radish, Celery, Asparagus, Basil, Parsley, Cilantro
225
+
226
+ Medical Procedure
227
+ Blood test, Vaccination, Colonoscopy, Appendectomy, Cesarean section, Vaginal delivery, Cataract surgery, Hip replacement, Knee replacement, Coronary angioplasty (PCI)
228
+
229
+ Affectionate Gestures
230
+ Hug, Kiss, Holding hands, Cuddling, Compliment, Love letter, Giving flowers, Romantic dinner, Surprise gift, Acts of service
231
+
232
+ Mobile OS
233
+ Android, iOS, HarmonyOS, Amazon Fire OS, KaiOS, Tizen, YunOS (AliOS), Series 40, Symbian, BlackBerry OS
234
+
235
+ Medical Specialty
236
+ Family Medicine, Internal Medicine, Pediatrics, Obstetrics and Gynecology, Emergency Medicine, Psychiatry, General Surgery, Cardiology, Neurology, Dermatology
237
+
238
+ Health Topic
239
+ Mental Health, Cardiovascular Disease, Cancer, Infectious Diseases, Diabetes, Respiratory Diseases, Nutrition, Physical Fitness, Women's Health, Substance Use Disorders
240
+
241
+ Eating Occasion
242
+ Breakfast, Lunch, Dinner, Snack, Brunch, Dessert, Late-night snack, On-the-go meal, Business meal, Holiday meal
243
+
244
+ Storage Type
245
+ Warehouse, Shipping container, Box, Pallet, Bin, Shelf, Refrigerator, Freezer, Tank, Cabinet
246
+
247
+ Coffee Beverage / Preparation Style
248
+ Espresso, Filter (Drip), Latte, Cappuccino, Americano, Iced Coffee, French Press, Cold Brew, Mocha, Instant
249
+
250
+ Environmental Impact Categories
251
+ Greenhouse Gas Emissions (CO2e), Energy Consumption, Water Use, Waste Generation, Air Pollution, Water Pollution, Land Use, Biodiversity Loss / Habitat Loss, Resource Depletion, Chemical Pollution / Toxicity
252
+
253
+ Game Genre
254
+ Action, Shooter, Role-Playing, Adventure, Puzzle, Sports, Strategy, Simulation, Racing, Horror
255
+
256
+ Holiday Name
257
+ Christmas Day, New Year's Day, New Year's Eve, Lunar New Year, Ramadan, Easter, Halloween, Valentine's Day, Eid al-Fitr, Diwali, Eid al-Adha, Hanukkah, Thanksgiving (United States), Mother's Day, Father's Day, Labor Day, St. Patrick's Day, Good Friday, Passover, Rosh Hashanah, Yom Kippur, Holi, Nowruz, Thanksgiving (Canada), Independence Day (United States), Memorial Day (United States), Veterans Day (United States), Canada Day, Bastille Day, Boxing Day, Cinco de Mayo, Kwanzaa
258
+
259
+ Streaming Service
260
+ Netflix, YouTube, Amazon Prime Video, Disney+, Hulu, Max, Apple TV+, Paramount+, Peacock, Pluto TV
261
+
262
+ Weather Type
263
+ Clear/Sunny, Cloudy/Partly Cloudy, Rain/Showers/Drizzle, Thunderstorm, Fog/Mist/Haze, Snow, Sleet/Freezing Rain, Windy, Hail, Dust/Sand
264
+
265
+ Cooking Technique
266
+ Baking, Boiling/Simmering, Frying, Grilling, Roasting, Steaming, Sautéing/Stir-frying, Braising/Stewing, Smoking, Pressure cooking
267
+
268
+ Religious Affiliation
269
+ Christianity, Islam, Hinduism, Buddhism, Judaism, Sikhism, Unaffiliated (Atheist/Agnostic/No religion), Indigenous/Traditional Religions, Other
270
+
271
+ Sentiment
272
+ Positive, Negative, Neutral, Mixed/Ambivalent, Support/Agree, Oppose/Disagree, Sarcastic, Factual, Unknown/Not Applicable
273
+
274
+ Biome Type
275
+ Tropical Rainforest, Temperate Forest, Boreal Forest (Taiga), Grassland/Savanna, Desert, Tundra, Mediterranean Shrubland (Chaparral), Wetlands (Marsh/Swamp/Peatland), Freshwater (Lakes/Rivers), Marine/Coastal (Reef/Estuary/Mangrove)
276
+
277
+ Playground Feature
278
+ Slide, Swing, Climbing structure (frame/wall/monkey bars), Seesaw, Sandbox, Roundabout/Merry-go-round, Playhouse, Balance elements (beam/stepping stones), Zip line, Trampoline
279
+
280
+ Handicraft Category
281
+ Sewing, Knitting, Crochet, Embroidery, Quilting, Jewelry making, Pottery/Ceramics, Woodworking, Leatherwork, Papercraft
282
+
283
+ Retail Location Type
284
+ Online / E-commerce, Downtown / Central Business District, Suburb, Main Street / High Street, Shopping Mall, Strip Mall / Strip Center, Outlet Center, Transit Hub / Airport Retail, Tourist / Entertainment District, Pop-up / Temporary Retail
285
+
286
+ Art and Craft Techniques
287
+ Painting, Drawing/Illustration, Photography, Sculpture, Printmaking, Ceramics, Digital Art, Textile Arts (weaving/embroidery), Woodworking/Carving, Jewelry/Metalworking
288
+
289
+ Transportation Infrastructure Type
290
+ Roads and Highways, Bridges, Railways, Airports, Ports (Seaports and Harbors), Public Transit (Buses, Trams, Metro), Tunnels, Bicycle Infrastructure (Bike Lanes, Cycle Paths), Pedestrian Infrastructure (Sidewalks, Footpaths), Intermodal Terminals (Freight)
291
+
292
+ Physical Activity
293
+ Walking, Running, Cycling, Swimming, Strength Training, Yoga, Team Sports, Dance, Hiking, Martial Arts
294
+
295
+ Pet Category
296
+ Dogs, Cats, Fish, Birds, Reptiles, Small Mammals, Amphibians
297
+
298
+ Clothing Type
299
+ Tops (T-shirt, Shirt, Blouse), Bottoms (Jeans, Pants, Shorts, Skirt, Leggings), Dresses, Outerwear (Jacket, Coat, Blazer), Sweaters (Sweater, Hoodie, Cardigan), Underwear, Socks, Shoes (Shoes, Boots, Sandals)
300
+
301
+ Dance Discipline
302
+ Ballet, Hip Hop, Contemporary, Jazz, Ballroom, Salsa, Tap, Breaking, Swing, Cultural/Street Styles (e.g., Afrobeats, K-Pop)
303
+
304
+ Musical Instruments
305
+ Piano, Guitar, Drums, Violin, Flute, Saxophone, Trumpet, Cello, Clarinet, Synthesizer
306
+
307
+ Building Type
308
+ Residential, Commercial, Office building, Retail, Mixed-use building, Industrial, Warehouse / Distribution, Hotel / Motel, Education (School/University), Healthcare (Hospital/Clinic)
309
+
310
+ Household Appliance
311
+ Refrigerator, Washing Machine, Dryer, Microwave, Oven / Cooktop, Dishwasher, Vacuum Cleaner, Television, Air Conditioner, Water Heater
312
+
313
+ Incident Type
314
+ Theft, Burglary, Robbery, Assault, Domestic Violence, Sexual Assault, Traffic Accident, Fraud, Vandalism, Drug Offense
315
+
316
+ Occupation
317
+ Healthcare (Nurse, Physician), Education (Teacher), Technology (Software Engineer), Business/Administration (Accountant, Administrative Assistant, Manager), Sales/Service (Retail Salesperson, Customer Service Representative, Cashier), Skilled Trades (Electrician, Plumber), Transportation/Logistics (Truck Driver), Construction (Construction Worker), Public Safety (Police Officer), Hospitality/Food (Cook)
318
+
319
+ Milestone Type
320
+ Birth, Graduation, Wedding, New Job, Promotion, Home Purchase, Retirement, Company Founded, Product Launch, Acquisition
321
+
322
+ Cyberattack Type
323
+ Phishing, Malware, Ransomware, Business Email Compromise (BEC), Distributed Denial of Service (DDoS), SQL Injection, Man-in-the-Middle (MITM), Credential Stuffing, Zero-Day Exploit, Insider Threat
324
+
325
+ Art Subject
326
+ Portrait, Landscape, Still Life, Abstract, Figurative, Cityscape, Seascape, Architecture, Floral, Wildlife, Historical, Religious, Mythological, Genre Scene, Self-Portrait, Nude, Conceptual, Fantasy, Social/Political, Allegory, Illustration, Sports, Industrial, Interior, Group Portrait, Narrative
327
+
328
+ Food Item
329
+ Bread, Rice, Potato, Chicken, Eggs, Milk, Pizza, Salad, Burger, Pasta, Sandwich, Soup, Fries, Fish, Noodles, Sushi, Curry, Tacos, Ice Cream, Cake, Pancake, Steak, Burrito, Dumpling, Cheese, Cereal, Chocolate, Hot Dog, Sausage, Shrimp, Beans, Oatmeal, Yogurt, Fruit, Vegetable
330
+
331
+ Insect Common Name
332
+ Mosquito, Housefly, Ant, Honey bee, Cockroach, Butterfly, Moth, Ladybug, Beetle, Fruit fly, Grasshopper, Locust, Dragonfly, Praying mantis, Termite, Aphid, Flea, Bed bug, Cicada, Firefly, Earwig, Weevil, Crane fly, Silkworm, Stink bug, Horsefly, Leafcutter ant, Soybean looper, Japanese beetle, Damselfly
333
+
334
+ Insurance Type
335
+ Health Insurance, Auto Insurance, Homeowners Insurance, Life Insurance, Renters Insurance, Travel Insurance, Pet Insurance, Disability Insurance, Dental Insurance, Vision Insurance, Long-Term Care Insurance, Umbrella Insurance, Motorcycle Insurance, Boat Insurance, Flood Insurance, Commercial Insurance, Workers' Compensation Insurance, Professional Liability Insurance, Cyber Insurance, Condo Insurance, Landlord Insurance, Title Insurance, Mortgage Insurance, RV Insurance, Mobile Home Insurance, Crop Insurance, Event Insurance, Identity Theft Insurance, Builder's Risk Insurance, Surety Bonds, Personal Articles Insurance
336
+
337
+ Venue Seating Section
338
+ General Admission, Standing Room, Floor, Orchestra, Balcony, Bleachers, Lower Bowl, Upper Bowl, Suite, Mezzanine
339
+
340
+ Home Decor Category
341
+ Wall Art, Rugs, Throw Pillows, Lighting, Curtains & Drapes, Indoor Plants, Mirrors, Candles, Vases, Throws & Blankets, Decorative Objects, Clocks, Shelving & Wall Storage, Baskets, Picture Frames, Sculptures, Planters & Pots, Tabletop Decor & Centerpieces, Decorative Trays, Bookends, Home Fragrance, Decorative Boxes, Accent Furniture, Mantel Decor, Seasonal Decor, Outdoor Decor, Table Linens, Wall Panels & Decals, Doormats & Welcome Mats, Storage & Organization
342
+
343
+ Country / Organization
344
+ United States, China, India, Russia, United Kingdom, France, Germany, Japan, South Korea, Canada, Australia, Brazil, Mexico, Argentina, Chile, Colombia, Peru, Spain, Portugal, Italy, Netherlands, Belgium, Switzerland, Austria, Sweden, Norway, Denmark, Finland, Poland, Czech Republic, Hungary, Romania, Bulgaria, Greece, Turkey, Israel, Saudi Arabia, United Arab Emirates, Qatar, Egypt, Morocco, South Africa, Nigeria, Kenya, Ethiopia, Ghana, Senegal, Tunisia, Algeria, Iran, United Nations, World Health Organization, World Bank, International Monetary Fund, World Trade Organization, European Union, African Union, ASEAN, NATO, OECD, UNICEF, UNESCO, UNHCR, International Red Cross, Amnesty International, Human Rights Watch, World Economic Forum, International Olympic Committee, FIFA, International Basketball Federation, Asian Development Bank, African Development Bank, European Central Bank, Federal Reserve, Asian Infrastructure Investment Bank, International Energy Agency, OPEC, International Atomic Energy Agency, Interpol, International Telecommunication Union, World Meteorological Organization, International Labour Organization, International Maritime Organization, World Intellectual Property Organization, International Civil Aviation Organization, International Organization for Migration, Doctors Without Borders, Save the Children, Greenpeace, World Wildlife Fund, International Chamber of Commerce, Transparency International, Global Fund, Gavi, International Crisis Group, Council of Europe, Arab League, Organization of American States, Commonwealth of Nations, World Customs Organization
345
+
346
+ Computer Game
347
+ Minecraft, Fortnite, League of Legends, Dota 2, Counter-Strike, Valorant, World of Warcraft, Grand Theft Auto V, The Witcher 3: Wild Hunt, Red Dead Redemption 2, Elden Ring, Dark Souls, Cyberpunk 2077, Call of Duty: Modern Warfare, Apex Legends, PUBG: Battlegrounds, Overwatch, StarCraft II, Diablo III, Hearthstone, The Elder Scrolls V: Skyrim, Fallout 4, Assassin’s Creed Valhalla, Assassin’s Creed Odyssey, Far Cry 5, Resident Evil 4, Monster Hunter: World, Final Fantasy XIV, The Legend of Zelda: Breath of the Wild, Super Mario Odyssey, Animal Crossing: New Horizons, Pokémon Red and Blue, Halo: Combat Evolved, Gears of War, God of War, Horizon Zero Dawn, Death Stranding, Metal Gear Solid V: The Phantom Pain, Sekiro: Shadows Die Twice, Civilization VI, Age of Empires II, The Sims 4, SimCity, Cities: Skylines, Kerbal Space Program, Terraria, Stardew Valley
348
+
349
+ City
350
+ New York City, Los Angeles, Chicago, Houston, Toronto, Vancouver, London, Paris, Berlin, Rome, Madrid, Barcelona, Amsterdam, Brussels, Zurich, Vienna, Stockholm, Copenhagen, Oslo, Helsinki, Tokyo, Osaka, Seoul, Beijing, Shanghai, Hong Kong, Singapore, Bangkok, Kuala Lumpur, Jakarta, Manila, Sydney, Melbourne, Auckland, Dubai, Abu Dhabi, Riyadh, Doha, Mumbai, Delhi, Bangalore, Chennai, Kolkata, São Paulo, Rio de Janeiro, Buenos Aires, Mexico City
351
+
352
+ Driving Environment
353
+ City, Highway, Suburban, Residential street, Intersection, Parking lot, Night driving, Wet road conditions, Construction zone, Snow or ice conditions
354
+
355
+ Fitness Goal
356
+ Weight Loss, General Fitness, Muscle Gain, Strength, Endurance, Flexibility, Mobility, Rehabilitation, Sports Performance, Weight Maintenance
357
+
358
+ Communication Channel
359
+ Website, Email, In-person, Phone call, Mobile app, SMS / Text message, Social media, Live chat, Messaging apps, Chatbot (automated chat), Push notification, Web form, Self-service portal, Knowledge base / FAQ, Video call, IVR / automated phone system, In-app messaging, Support ticketing system, Postal mail, Community forum, Kiosk, Voice assistant, Fax
360
+
361
+ Grain Type
362
+ Wheat, Rice, Corn, Barley, Oats, Rye, Sorghum, Millet, Quinoa, Buckwheat
363
+
364
+ Furniture Category
365
+ Chair, Table, Sofa, Bed, Desk, Cabinet, Dresser, Wardrobe, Bookcase, Nightstand
366
+
367
+ Heritage Site Type
368
+ Museum, Historic Building, Historic District, Archaeological Site, Monument, Religious Site, Castle, Palace, Memorial, Cultural Landscape
369
+
370
+ Beverage Type
371
+ Water, Coffee, Tea, Soda, Juice, Milk, Beer, Wine, Spirits, Energy Drink, Sparkling Water, Smoothie, Lemonade, Iced Tea, Cold Brew Coffee, Kombucha, Cocktail, Hot Chocolate, Plant-based Milk, Herbal Tea, Milkshake, Sports Drink, Cider, Kefir, Coconut Water, Espresso, Iced Coffee, Flavored Milk, Sake
372
+
373
+ Research Publication Type
374
+ Journal Article, Review Article, Preprint, Conference Paper, Book, Book Chapter, Thesis/Dissertation, Technical Report, Working Paper, Dataset, Letter / Short Communication, Conference Abstract, Editorial, Methods / Protocol, Patent, Case Report, Conference Poster, Monograph, Book Review, White Paper, Policy Brief, Software / Code Release, Commentary, Standard (Technical Standard), Erratum / Correction, Retraction
375
+
376
+ Earring Type
377
+ Studs, Hoops, Dangles, Clip-On Earrings, Ear Cuffs, Chandeliers, Huggies, Threaders
378
+
379
+ Common Illicit and Recreational Drugs
380
+ Alcohol, Nicotine, Caffeine, Cannabis, Cocaine, Methamphetamine, MDMA, Heroin, Benzodiazepines, Fentanyl
381
+
382
+ Wearable Accessory Type
383
+ Sunglasses, Eyeglasses, Belt, Bag, Hat, Scarf, Ring, Earrings, Necklace, Watch, Smartwatch, Fitness tracker, Bracelet, Gloves, Tie, Headband, Anklet, Cufflinks, Brooch, Hair accessory, Locket, Pocket square
384
+
385
+ Play Activity Type
386
+ Free Play, Imaginative/Role Play, Physical Play, Outdoor Play, Social/Cooperative Play, Construction/Building Play, Arts & Crafts, Ball Play, Board Games, Puzzle Play, Sensory Play, Water Play, Nature Exploration, Team/Organized Sports, Digital/Screen Play, Music and Movement, STEM Play, Manipulative/Fine Motor Play, Rough-and-Tumble Play, Independent/Solitary Play, Exploratory/Discovery Play, Card Games, Chasing/Tag Games, Creative Storytelling
387
+
388
+ Music Distribution Format
389
+ On-demand streaming, Terrestrial radio, Digital download, Physical media (CD/Vinyl/Cassette), Music video, Live performance, Short-form social media clips, Satellite radio, Ringtone, Sheet music
390
+
391
+ Sports Court Surface
392
+ Concrete, Asphalt, Hardwood, Grass, Artificial turf, Clay, Sand, Rubber, Carpet, Modular plastic tiles
393
+
394
+ Application Category (App Store)
395
+ Social Networking, Games, Entertainment, Productivity, Utilities, Finance, Health & Fitness, Shopping, Communication, Photo & Video, Music, Education, Travel, Food & Drink, News, Sports, Business, Navigation, Lifestyle, Weather, Books, Reference, Medical, Family, Events, Dating, Home Automation, Art & Design
396
+
397
+ Exercise Type
398
+ Walking, Running, Strength Training, Yoga, Cycling, Indoor Cycling, Weightlifting, High-Intensity Interval Training, Swimming, Pilates, CrossFit, Calisthenics, Zumba, Boxing, Kickboxing, Barre, Dance Fitness, Circuit Training, Rowing, Hiking, Aerobics, Mobility/Stretching, Martial Arts, Tai Chi, Functional Training, Climbing, Powerlifting, Olympic Weightlifting, Sprint Training, Stair Climbing
399
+
400
+ Streaming Platforms
401
+ Netflix, YouTube, Amazon Prime Video, Disney+, Hulu, Twitch, Spotify, Apple Music, Max, Paramount+
402
+
403
+ Cultivation Environment
404
+ Open Field, Backyard Garden, Container Gardening, Greenhouse, Plant Nursery, Raised Bed, Orchard, Hydroponics, Vertical Farm, Aquaponics
405
+
406
+ Sensor Type
407
+ Camera, Accelerometer, Microphone, Temperature Sensor, GNSS / GPS Receiver, Ambient Light Sensor, Capacitive Touch Sensor, Gyroscope, Magnetometer, Pressure Sensor, Humidity Sensor, Voltage Sensor, Current Sensor, Proximity Sensor, Ultrasonic Sensor, Radar, LiDAR, Thermal Imaging Camera, Vibration Sensor, Gas Sensor, CO2 Sensor, Particulate Matter (PM) Sensor, Force Sensor, Flow Sensor, Level Sensor, pH Sensor, Hall Effect Sensor, Inertial Measurement Unit (IMU)
408
+
409
+ Wine Grape Varieties and Styles
410
+ Cabernet Sauvignon, Chardonnay, Merlot, Pinot Noir, Sauvignon Blanc, Syrah/Shiraz, Riesling, Malbec, Rosé, Sparkling
411
+
412
+ Manufacturing Step
413
+ Material Procurement, Machining, Forming, Casting, Welding, Surface Treatment, Assembly, Quality Inspection, Testing, Packaging
414
+
415
+ Amusement Ride Type
416
+ Roller Coaster, Ferris Wheel, Carousel, Bumper Cars, Water Ride, Drop Tower, Dark Ride, Motion Simulator, Teacups, 4D Cinema
417
+
418
+ Dish Type
419
+ Pizza, Burger, Pasta, Salad, Sushi, Steak, Ramen, Taco, Sandwich, Soup, Fried Chicken, Curry, Noodles, Barbecue, Seafood, Dumplings, Wrap, Stir-fry, Kebab, Dim Sum, Paella, Risotto, Omelette, Pancake, Hot Pot, Sashimi, Tapas, Dessert
420
+
421
+ Retailer Name
422
+ Amazon, Walmart, Alibaba, eBay, Costco, Kroger, Target, The Home Depot, Walgreens, CVS Pharmacy, Tesco, Carrefour, IKEA, Lidl, Aldi, Best Buy, Macy's, Nordstrom, H&M, Zara, Sephora, Ulta Beauty, Dollar General, Dollar Tree, 7-Eleven, Sam's Club, JD.com, Flipkart, Mercado Libre, Coupang
423
+
424
+ Pesticide Type (Target/Function)
425
+ Herbicide, Insecticide, Fungicide, Rodenticide, Biopesticide, Acaricide, Nematicide, Fumigant, Repellent, Plant Growth Regulator
426
+
427
+ Vehicle Maintenance Type
428
+ Oil Change, Tire Rotation, Tire Replacement, Brake Inspection, Brake Pad Replacement, Battery Test & Replacement, Wheel Alignment, Fluid Check & Top-off, Engine Diagnostic (OBD Scan), A/C Service & Refrigerant Recharge
429
+
430
+ Painting Medium
431
+ Acrylic, Oil, Watercolor, Digital painting, Soft pastel, Gouache, Mixed media, Ink, Spray paint, Airbrush, Oil pastel, Egg tempera, Encaustic, Casein, Alcohol ink, Enamel, Charcoal, Collage, Metalpoint, Water‑mixable oil, Acrylic ink, Acrylic gouache, Marker, Oil stick, Gilding
432
+
433
+ Craft Supplies
434
+ Paper, Glue, Paint, Scissors, Yarn, Fabric, Markers, Pencils, Brushes, Canvas, Cardstock, Stickers, Adhesive tape, Hot glue gun, Glue sticks, Beads, Buttons, Ribbon, Wire, Clasps, Clay, Felt, Foam sheets, Pipe cleaners, Sequins, Glitter, Stencils, Stamps, Ink pads, Embroidery floss, Needles, Googly eyes, Craft knife, Cutting mat
435
+
436
+ Project Phase
437
+ Initiation, Planning, Design, Procurement, Construction, Testing, Commissioning, Handover, Operation, Closeout
438
+
439
+ Dominant Forest Cover Type
440
+ Tropical Rainforest, Temperate Rainforest, Deciduous Forest, Coniferous Forest, Boreal Forest (Taiga), Mixed Forest, Mangrove Forest, Woodland, Plantation Forest, Riparian Forest
441
+
442
+ Festival Theme
443
+ Music, Food, Art, Film, Cultural, Street Fair, Beer, Wine, Family, Parade
444
+
445
+ Camera Lens Type
446
+ Zoom lens, Prime lens, Wide-angle lens, Telephoto lens, Macro lens, Fisheye lens, Tilt-shift lens, Pancake lens
447
+
448
+ Concession Stand Product
449
+ Popcorn, Soda, Candy, Bottled Water, Beer, Nachos, Pretzel, Hot Dog, Pizza Slice, Ice Cream
450
+
451
+ Climate Classification
452
+ Tropical, Arid, Temperate, Continental, Mediterranean, Oceanic (Maritime), Subtropical, Polar
453
+
454
+ Sports Facility Type
455
+ Gym / Fitness Center, Soccer Field / Football Pitch, Basketball Court, Swimming Pool, Tennis Court, Running Track / Athletics Track, Baseball Field, Indoor Arena, Golf Course, Ice Rink
456
+
457
+ Finishing Position
458
+ First, Second, Third, Top 10, Finalist, Semi-finalist, Quarter-finalist, Did Not Finish, Disqualified, Tied
459
+
460
+ Sustainability Category
461
+ Climate Change, Renewable Energy, Energy Efficiency, Waste Management, Recycling, Water Management, Sustainable Agriculture, Sustainable Transportation, Biodiversity, Circular Economy
462
+
463
+ Audio Listening Method
464
+ Subscription streaming, Ad-supported streaming, Terrestrial radio (AM/FM), Podcasts, Internet radio (streamed radio stations), Local files (stored audio: MP3, FLAC), Purchased digital downloads, Physical CDs, Vinyl records, Casting / AirPlay / Bluetooth (device-to-speaker)
465
+
466
+ Travel Market Segment
467
+ Leisure, Business, Family, Budget, Luxury, Solo, Adventure, Romantic/Honeymoon, Cruise, Meetings & Events (MICE)
468
+
469
+ Occasion (Usage Scenario)
470
+ Everyday / Daily Wear, Work / Office, Casual Outing, Travel / Vacation, Party, Wedding, Formal Event, Job Interview, Holiday (e.g., Christmas/New Year's), Funeral
471
+
472
+ Major Animal Groups
473
+ Mammals, Birds, Fish, Reptiles, Amphibians, Insects, Arachnids, Mollusks, Crustaceans, Worms
474
+
475
+ Alcoholic Beverage Type
476
+ Beer, Wine, Whiskey, Vodka, Rum, Gin, Tequila, Champagne / Sparkling Wine, Cocktail, Sake
477
+
478
+ Establishment Type
479
+ Restaurant, Cafe / Coffee Shop, Bar / Pub, Grocery Store / Supermarket, Convenience Store, Retail Store, Hotel, Hospital / Clinic, School, Bank Branch
480
+
481
+ Website Content Section
482
+ News, Sports, Business, Technology, Entertainment, Lifestyle, Health, Politics, Travel, Opinion
483
+
484
+ Property Flooring Type
485
+ Hardwood, Carpet, Tile, Vinyl, Laminate, Concrete, Natural stone, Bamboo, Cork, Linoleum
486
+
487
+ Vehicle Propulsion Type
488
+ Gasoline (Petrol), Diesel, Battery Electric (BEV), Hybrid Electric (HEV), Plug-in Hybrid Electric (PHEV), CNG (Compressed Natural Gas), LPG (Liquefied Petroleum Gas), Fuel Cell Electric (Hydrogen), Ethanol / Flex-Fuel, Jet Engine / Turbine
489
+
490
+ Retail Store Category
491
+ Grocery, Convenience Store, Department Store, Discount Store, Pharmacy, Clothing Store, Electronics Store, Home Improvement Store, E-commerce / Online Retailer, Furniture Store
492
+
493
+ Email Service Provider
494
+ Gmail, Outlook.com, Yahoo Mail, iCloud Mail, ProtonMail, Yandex Mail, Zoho Mail, Custom/Corporate Email
495
+
496
+ Costume Character
497
+ Witch, Superhero, Princess, Vampire, Ghost, Zombie, Pirate, Clown, Wizard, Ninja
498
+
499
+ Livestock Species
500
+ Cattle, Pig, Chicken, Sheep, Goat, Aquaculture, Duck, Turkey, Rabbit, Bees
501
+
502
+ Historical Period
503
+ Prehistory, Ancient Egypt, Ancient Greece, Ancient Rome, Medieval Period, Renaissance, Industrial Revolution, World War I, World War II, Cold War
504
+
505
+ Class Type
506
+ Yoga, Pilates, Strength Training, High-Intensity Interval Training, Cardio, Spinning, Zumba, Cooking, Baking, Dance
507
+
508
+ Religious Symbol
509
+ Cross, Crescent and Star, Star of David, Om (Aum), Yin Yang, Dharma Wheel, Menorah, Hamsa, Ankh, Lotus
510
+
511
+ Fruit Name
512
+ Banana, Apple, Orange, Grape, Strawberry, Watermelon, Mango, Pineapple, Lemon, Pear, Blueberry, Cherry, Avocado, Grapefruit, Kiwi, Papaya, Plum, Peach, Apricot, Pomegranate, Raspberry, Coconut, Cantaloupe, Honeydew, Fig, Date, Lychee, Jackfruit, Passionfruit, Guava, Persimmon, Cranberry, Mandarin, Starfruit, Kumquat
513
+
514
+ Pollution Source
515
+ Transportation, Industrial Sources, Municipal Sewage and Wastewater, Agricultural Runoff, Stormwater and Urban Runoff, Landfills and Solid Waste Disposal, Residential Household Waste, Energy Production, Oil and Gas Operations, Plastic Waste and Marine Debris, Mining and Extractive Activities, Shipping and Ports, Accidental Chemical Spills, Hazardous Waste Sites, Pesticide and Fertilizer Use, Livestock Operations, Atmospheric Deposition, Construction and Demolition, Forestry and Logging, Fishing and Aquaculture, Industrial Air Emissions, Road Salt and De-icing, Illegal Dumping and Littering, Nonpoint Source Pollution, Point Source Discharges, Natural Sources, Tourism and Recreational Activities
516
+
517
+ Dietary Food Group
518
+ Vegetables, Fruits, Grains, Protein foods, Dairy, Fats and oils, Beverages, Sugars and sweets, Processed foods, Snacks
519
+
520
+ Neighborhood Name
521
+ Downtown, Midtown, Uptown, Suburb, Old Town, Historic District, Financial District, University District, Arts District, Waterfront
522
+
523
+ Competition Gender Division
524
+ Men, Women, Mixed, Open, Boys, Girls, Non-binary, Unspecified, Other
525
+
526
+ Irrigation Source Type
527
+ Groundwater, Surface water, Canal, Rainwater harvesting, Treated wastewater (recycled/effluent), Municipal / tap water, Desalinated water, Stormwater / surface runoff, Other / unspecified
528
+
529
+ Gift Category
530
+ Gift Cards, Clothing & Apparel, Electronics, Beauty & Personal Care, Home Decor, Books, Food & Gourmet, Toys & Games, Experiences (tickets, classes, travel), Flowers & Plants
531
+
532
+ Vessel Type
533
+ Container Ship, Bulk Carrier, Tanker, General Cargo Ship, Fishing Vessel, Ferry, Tug, Cruise Ship, Barge, Naval Vessel
534
+
535
+ Online Video Platforms
536
+ YouTube, TikTok, Netflix, Facebook Watch, Instagram, Amazon Prime Video, Disney+, Tencent Video, Twitch, Hulu, HBO Max, Apple TV+, iQIYI, Disney+ Hotstar, Snapchat, Bilibili, Youku, Tubi, Dailymotion, Vimeo, Peacock, Paramount+, Pluto TV, Roku Channel, Sling TV, FuboTV, ESPN+, DAZN, Crunchyroll, Rakuten Viki, Mubi, Shudder, CuriosityStream, Kanopy, Crackle, Viu
537
+
538
+ Tropical Cyclone Intensity Category (Saffir–Simpson and related classifications)
539
+ Tropical Depression, Tropical Storm, Category 1, Category 2, Category 3, Category 4, Category 5, Extratropical Cyclone, Post-Tropical Cyclone
540
+
541
+ Internet Connectivity Status
542
+ Online, Offline, Limited Connectivity, Intermittent Connectivity, Connecting, Connection Failed, Authentication Required, Captive Portal, Throttled, Maintenance
543
+
544
+ Water Supply Type
545
+ Piped household connection, Public tap / standpipe, Packaged / bottled water, Vendor-delivered water (tanker / cart), Borehole (drilled well), Dug well, Surface water (river / lake / pond / stream), Rainwater harvesting, Spring, Other / unspecified source
546
+
547
+ Performance Terrain Type
548
+ Road, Trail, Gravel Road, Dirt Road, Rocky Terrain, Sand / Beach, Mud, Snow / Ice, Grass, Mixed Terrain
549
+
550
+ Software Application
551
+ Web Browser, Email, Messaging/Chat, Social Media, Video Conferencing, Office Productivity, Cloud Storage, Streaming Media, Design/Photo Editing, Learning/Education
552
+
553
+ Result Status (Test/Operation)
554
+ Pass, Fail, Skipped, Not Run, Error, Pending, Running, Cancelled, Timed Out, Inconclusive
555
+
556
+ Weightlifting and Strength Training Exercises
557
+ Squat, Bench Press, Deadlift, Pull-up, Overhead Press, Row, Lunge, Plank, Bicep Curl, Kettlebell Swing
558
+
559
+ Access Level
560
+ No Access, Guest, Read-Only, Editor (Read/Write), Administrator, Owner
561
+
562
+ Therapeutic Area
563
+ Oncology, Cardiology, Infectious Diseases, Neurology, Psychiatry, Endocrinology, Respiratory, Gastroenterology, Dermatology, Rheumatology, Nephrology, Hematology, Pediatrics, Obstetrics & Gynecology, Allergy/Immunology, Pain Management, Ophthalmology, Otolaryngology, Geriatrics, Orthopedics, Emergency Medicine, Critical Care, Transplantation, Rare Diseases, Addiction Medicine, Genetics & Genomic Medicine, Public Health & Preventive Medicine, Vaccines, Sleep Medicine, Sports Medicine, Dentistry
564
+
565
+ Employment Type
566
+ Full-time, Part-time, Permanent, Temporary, Contract (Fixed-term), Independent Contractor, Internship, Seasonal, Remote, Hybrid
567
+
568
+ Event Category
569
+ Conference, Concert, Sports Event, Wedding, Trade Show, Exhibition, Workshop, Meetup, Webinar, Fundraiser
570
+
571
+ Wellness Program Type
572
+ Fitness Classes, Health Screenings, Nutrition Counseling, Mental Health Counseling, Employee Assistance Program, Smoking Cessation, Weight Management, Stress Management, Health Coaching, Wellness Challenges
573
+
574
+ Packaging Material
575
+ Plastic, Paper, Cardboard, Glass, Aluminum, Steel, Wood, Composite (Laminate), Bioplastic, Foam
576
+
577
+ ADAS Feature
578
+ Rear View Camera, Automatic Emergency Braking, Lane Departure Warning, Lane Keeping Assist, Adaptive Cruise Control, Blind Spot Monitoring, Parking Sensors, Forward Collision Warning, Pedestrian Detection, Traffic Sign Recognition
579
+
580
+ Workout Modality
581
+ Walking, Running, Strength Training, Yoga, Cycling, Swimming, HIIT, Pilates, Hiking, Martial Arts
582
+
583
+ Common Pest Type
584
+ Ants, Mosquitoes, Flies, Cockroaches, Rodents, Termites, Bed Bugs, Spiders, Ticks, Fleas
585
+
586
+ Patrol Method
587
+ Foot, Vehicle (Patrol car), Bicycle, Motorcycle, Mobile (roving) patrol, Static (Fixed-post / Checkpoint), Remote video surveillance (CCTV), Drone (UAV), K9 (Canine), Community / Neighborhood watch
588
+
589
+ Sustainable Building Feature
590
+ LED lighting, High-quality insulation, High-performance windows, Energy-efficient HVAC systems, Solar panels, Heat pumps, Smart energy management systems, Water-efficient fixtures, Low-VOC materials, EV charging stations
591
+
592
+ Fatal Incident Type
593
+ Motor vehicle collision, Fall, Suicide (intentional self-harm), Poisoning / drug overdose, Drowning, Homicide / assault, Structure fire, Industrial / workplace accident, Aviation accident, Electrocution
594
+
595
+ Learning Resource Type
596
+ Videos, Articles, Online Courses, Tutorials, E-books, Podcasts, Textbooks, Webinars, Documentation, Forums / Discussion Threads
597
+
598
+ Service Category
599
+ Electricity, Water, Internet & Telecommunications, Waste Management, Transportation, Healthcare Services, Emergency Services, Education Services, Sanitation, Natural Gas
600
+
601
+ Manufacturing Process
602
+ Assembly, Machining, Cutting, Welding, Injection Molding, Casting, Forging, Stamping, Extrusion, Additive Manufacturing
603
+
604
+ Pest Control Technique
605
+ Integrated Pest Management (IPM), Chemical control (pesticides), Biological control, Trapping, Exclusion / Barriers, Baiting, Monitoring / Surveillance, Habitat modification, Fumigation, Manual removal
606
+
607
+ Personal Financial Concern Category
608
+ Day-to-day living expenses, Housing affordability, Income loss / unemployment, High consumer debt (credit card debt), Unexpected medical bills, Retirement savings shortfall, Inflation and rising prices, Insufficient emergency fund, Student loan debt, Insurance costs (health, auto, home, life)
609
+
610
+ Military Platform Type
611
+ Fighter Aircraft, Transport Aircraft, Attack Helicopter, Main Battle Tank, Armored Personnel Carrier (APC), Destroyer, Frigate, Submarine, Aircraft Carrier, Self-Propelled Artillery
612
+
613
+ Subscription Plan Tier
614
+ Free, Free Trial, Basic, Standard, Premium, Business, Enterprise, Student, Family
615
+
616
+ Organization Sector
617
+ Healthcare & Medical, Education, Government / Public Sector, Retail & E-commerce, Finance & Banking, Information Technology (IT), Manufacturing, Construction, Real Estate, Transportation & Logistics, Professional Services, Nonprofit / NGO, Agriculture & Forestry, Hospitality & Tourism, Media & Entertainment, Energy & Utilities, Legal Services, Telecommunications, Insurance, Pharmaceuticals & Biotechnology, Arts & Culture, Environmental & Conservation, Religious / Faith-based, Sports & Recreation, Research & Development, Defense & Aerospace, Mining & Extraction, Consumer Goods / FMCG, Wholesale & Distribution
618
+
619
+ Project Category
620
+ New Construction, Renovation / Remodeling, Residential, Commercial, Infrastructure, Site Development, Demolition, Adaptive Reuse, Landscaping, Environmental Restoration
621
+
622
+ Footwear Style
623
+ Sneakers, Boots, Sandals, Heels, Flats, Loafers, Oxfords, Slippers, Flip-flops, Running Shoes
624
+
625
+ Forage Type
626
+ Pasture, Grass Hay, Alfalfa, Silage, Haylage, Straw, Rangeland, Clover, Crop Residues
627
+
628
+ Clinical Service Type
629
+ Primary Care Visit, Telehealth Visit, Preventive Care, Lab Services, Diagnostic Imaging, Medication Management, Emergency Department Visit, Surgical Consultation, Physical Therapy, Individual Therapy
630
+
631
+ Hat Style
632
+ Baseball cap, Beanie, Bucket hat, Sun hat, Fedora, Cowboy hat, Beret, Flat cap, Visor, Straw hat
633
+
634
+ Technology Solution Type
635
+ Sensors, Cloud Services, IoT Platforms, Edge Computing, AI/ML Solutions, Remote Monitoring, SCADA Systems, Energy Management Systems, Building Management Systems, EV Charging Stations
636
+
637
+ Roofing Material/Type
638
+ Asphalt Shingles, Metal Roofing, Clay Tile, Concrete Tile, Built-Up Roofing, EPDM Rubber Roofing, TPO Roofing, Modified Bitumen, PVC Roofing, Slate, Wood Shingles, Wood Shakes, Composite Shingles, Synthetic Slate, Solar Tiles, Spray Polyurethane Foam, Green Roof, Stone-Coated Metal, Standing Seam Metal, Corrugated Metal, Thatch
639
+
640
+ Insect Group (common names)
641
+ Ant, Bee, Beetle, Butterfly, Moth, Mosquito, Fly, Wasp, Cockroach, Termite
642
+
643
+ Movie Theater Format
644
+ Standard 2D, 3D, IMAX, Dolby Cinema, Premium Large Format, Drive-In, Art-house Cinema, 4DX, Outdoor/Open-Air Cinema, Luxury Seating
645
+
646
+ Food Item Name
647
+ Milk, Bread, Eggs, Bottled Water, Rice, Pasta, Chicken Breast, Ground Beef, Cheese, Butter, Yogurt, Frozen Vegetables, Potatoes, Apples, Bananas, Tomatoes, Onions, Lettuce, Cooking Oil, Sugar, Salt, Coffee, Soda, Orange Juice, Ketchup, Chocolate Chip Cookie, Ice Cream, Potato Chips, Hamburger Bun, Coleslaw, Iced Tea, Lemonade, Baked Beans, Ground Turkey, Pasta Salad, Watermelon, Cheddar, Mayonnaise, Flour
648
+
649
+ Medical Treatment Type
650
+ Medication, Vaccination, Preventive care, Surgery, Physical therapy, Psychotherapy, Lifestyle intervention, Radiation therapy, Chemotherapy, Immunotherapy, Palliative care, Emergency care, Medical device therapy, Transplantation, Dialysis, Blood transfusion, Hormone therapy, Occupational therapy, Speech therapy, Cognitive behavioral therapy, Behavioral intervention, Nutritional therapy, Dietary supplement, Complementary and alternative medicine, Acupuncture, Hospice care, Pain management, Gene therapy, Stem cell therapy, Placebo
651
+
652
+ Handbag Style
653
+ Tote, Shoulder bag, Crossbody bag, Backpack, Clutch, Messenger bag, Satchel, Hobo bag, Duffel bag, Briefcase, Belt bag, Wallet, Wristlet, Bucket bag, Top-handle bag, Saddle bag, Sling bag, Drawstring bag, Cosmetic bag, Weekender bag, Evening bag, Minaudière, Bowling bag, Doctor bag, Camera bag, Laptop bag, Frame bag, Carryall
654
+
655
+ Educational Program Type
656
+ Undergraduate Degree Program, Graduate Degree Program, Certificate Program, Online Program, Professional Development, Internship, Apprenticeship, Study Abroad, Fellowship, Continuing Education
657
+
658
+ Art Installation Type
659
+ Sculpture, Public Art, Installation Art, Monument, Memorial, Site-specific Installation, Light Installation, Interactive Installation, Video Installation, Sound Installation
660
+
661
+ Tour Activity
662
+ Sightseeing tour, Walking tour, City tour, Cultural/heritage tour, Food tour, Hiking, Boat cruise, Cycling tour, Wildlife safari, Snorkeling
663
+
664
+ Server Role
665
+ Web Server, Database Server, Application Server, Load Balancer, File Server, Cache Server, Mail Server, DNS Server, Authentication / Directory Server, Proxy Server, Backup Server, Monitoring Server, Logging Server, CI/CD Server, Storage Server, VPN Server, FTP / SFTP Server, DHCP Server, NTP (Time) Server, Container Registry, Orchestration / Cluster Management Server, API Gateway, Edge / CDN Node, Print Server, Media Streaming Server, Certificate Authority (CA) Server, Analytics / Big Data Server, Game Server, Remote Desktop / Terminal Server, License Server
666
+
667
+ Emotion Type
668
+ Happiness, Sadness, Anger, Fear, Surprise, Disgust, Love, Anxiety, Trust, Anticipation, Guilt, Shame, Pride, Jealousy, Envy, Contempt, Boredom, Excitement, Relief, Awe, Hope, Grief, Nostalgia, Compassion, Embarrassment, Curiosity, Contentment, Frustration, Serenity, Loneliness, Desire, Interest
669
+
670
+ Cause of Damage
671
+ Water damage, Fire, Storm and wind damage, Theft, Plumbing failure (burst pipes/leaks), Flood (natural flooding), Impact / collision (vehicle, falling objects), Hail, Vandalism / malicious damage, Mold and mildew, Freeze / frost damage, Electrical damage / power surge, Explosion, Structural failure / collapse, Subsidence / ground movement, Pest infestation (e.g., termites, rodents), Corrosion / rust, Snow / ice load damage, Accidental damage (drops, spills, human error), Wear and tear / gradual deterioration, Chemical contamination / spill, Oil or fuel leak, Biological contamination (sewage, biohazard), Lightning strike, Heat / thermal damage (non-fire), Landslide / mudslide, Storm surge / coastal inundation, Glass breakage
672
+
673
+ Healthcare Facility Type
674
+ Primary Care Clinic, Pharmacy, Urgent Care Clinic, Community Health Center, General Hospital, Emergency Department, Outpatient Clinic, Ambulatory Surgery Center, Skilled Nursing Facility, Assisted Living Facility, Home Health Agency, Hospice, Rehabilitation Hospital, Behavioral Health Center, Diagnostic Laboratory, Imaging/ Radiology Center, Dialysis Center, Birthing Center, Children's Hospital, Teaching Hospital, Community Hospital, Critical Access Hospital, Long-Term Acute Care Hospital (LTACH), Trauma Center, VA Hospital, Mobile Clinic, Retail Clinic, Dental Clinic, Optometry / Eye Clinic, Blood Bank / Transfusion Center, Public Health Department, Sexual and Reproductive Health Clinic, Occupational Health Clinic, School Health Center, Telehealth Service, Outpatient Rehabilitation Center, Psychiatric Hospital
675
+
676
+ Application Functionality
677
+ Authentication / Login, Push Notifications, User Profile (Account Management), Payments / Checkout, Search, Maps & Location (GPS), Camera / Photo Capture, Offline Mode & Data Sync, Analytics & Reporting, In-app Messaging / Chat, Social Sharing, Security & Encryption, Image Recognition, Barcode / QR Scanning, Forms & Surveys, Scheduling / Calendar / Booking, File Upload / Download (Document Viewer), API Integration / Webhooks, E-commerce / Product Catalog, Video Conferencing / Live Streaming, Machine Learning Recommendations, Role-based Access Control (Permissions), Voice Calling / VoIP, Alerts & Alarms, Backup & Restore / Data Export, Customer Support / Helpdesk / Live Chat, Feedback, Ratings & Reviews, Inventory & Order Management, Gamification & Badges, Remote Monitoring & Telemetry, Augmented Reality (AR), Equipment Maintenance, Pest Identification
678
+
679
+ Ecosystem / Habitat Type
680
+ Forest, Grassland, Desert, Wetland, Freshwater, Marine, Agricultural land, Urban, Coastal
681
+
682
+ Craft Materials
683
+ Fabric, Paper, Yarn, Glue, Paint, Wood, Metal, Plastic/Resin, Leather, Wire
684
+
685
+ Subsystem / Component Type
686
+ Processor / Microcontroller, Sensor, Controller / Electronic Control Unit (ECU), Battery, Power Supply, Electric Motor, Inverter, Switch, Pump, Valve
687
+
688
+ Flavor Profile (Tasting Descriptors)
689
+ Sweet, Salty, Sour/Acidic, Bitter, Umami, Fruity, Spicy, Floral, Smoky, Nutty
690
+
691
+ Transport Route Type
692
+ Highway / Expressway, Local Road / Street, Arterial Road, Collector Road, Ramp (On/Off Ramp), Rail Line, Subway / Metro Line, Bus Route, Bicycle Route, Ferry Route
693
+
694
+ Basic Human Needs
695
+ Food and Nutrition, Clean Water, Housing, Healthcare, Sanitation, Education, Income and Livelihood, Safety and Security, Energy and Utilities, Transportation
696
+
697
+ Fashion Style
698
+ Casual, Formal, Business Casual, Streetwear, Athleisure, Vintage/Retro, Bohemian, Minimalist, Punk, Gothic
699
+
700
+ Humanitarian Aid Sector
701
+ Health, Food Security and Agriculture, Water, Sanitation and Hygiene (WASH), Shelter and Non-Food Items (NFI), Protection, Education, Nutrition, Cash and Voucher Assistance (CVA), Logistics, Coordination and Information Management
702
+
703
+ Character Type
704
+ Human, Animal, Robot/Android, Alien, Monster, Mythical Creature, Undead, Deity/Demon, Ghost/Spirit, Cyborg
705
+
706
+ Aircraft Category (type/market segment)
707
+ Airplane (fixed-wing), Helicopter (rotorcraft), Narrow-body airliner, Wide-body airliner, Business jet, Regional aircraft, Cargo aircraft (freighter), Military aircraft, Glider (sailplane), VTOL / eVTOL
708
+
709
+ Benefit Recipient Type
710
+ Children, Seniors, People with disabilities, Low-income households, Unemployed individuals, Refugees and asylum seekers, Homeless individuals, Veterans, Students, Indigenous peoples
711
+
712
+ Forms of Folklore
713
+ Myth, Legend, Folktale, Fairy tale, Fable, Proverb, Riddle, Folk song, Urban legend, Superstition
714
+
715
+ Generational Cohort
716
+ Generation Alpha, Generation Z, Millennials, Generation X, Baby Boomers, Silent Generation, Greatest Generation
717
+
718
+ Vehicle Part Type
719
+ Engine, Transmission, Brakes, Tires, Battery, Suspension, Steering, Electrical System, Fuel System, Exhaust System
720
+
721
+ Ad Format
722
+ Search Ads, Social Media Ads, Display (Banner) Ads, Video Ads, Email Marketing, In-App Ads, Native Ads, Connected TV (CTV) Ads, Audio Ads, Influencer Marketing
723
+
724
+ Consumer Electronics & Computer Hardware Category
725
+ Smartphones, Headphones & Earbuds, Laptops, Televisions, Tablets, Desktop Computers, Monitors, Smartwatches, Smart Speakers, Game Consoles, Printers & Scanners, Keyboards, Computer Mice, External Storage Devices, Storage Drives (HDDs & SSDs), Networking Equipment (Routers, Modems, Switches), Graphics Cards, CPUs (Processors), Digital Cameras, Streaming Media Players, Smart Home Devices, Security Cameras & Surveillance, Projectors, Power Banks, Network Attached Storage (NAS), Drones, Virtual Reality & Augmented Reality Headsets, Fitness Trackers, E-Readers, Motherboards
726
+
727
+ U.S. Military Service Branch
728
+ Army, Navy, Air Force, Marine Corps, Space Force, Coast Guard, National Guard, Merchant Marine
729
+
730
+ App Permission
731
+ Internet / Network Access, Location, Camera, Microphone, Storage / Files & Media, Contacts, Notifications / Push, Bluetooth, SMS, Biometric authentication (FaceID/TouchID)
732
+
733
+ News Organization
734
+ Associated Press, Reuters, BBC News, CNN, The New York Times, The Washington Post, The Wall Street Journal, Fox News, Al Jazeera, Bloomberg
735
+
736
+ Construction & Infrastructure Project Type
737
+ Residential, Commercial, Industrial, Roads & Highways, Bridges, Rail & Public Transit, Airports, Ports & Harbors, Water & Wastewater, Power & Energy
738
+
739
+ Home Feature
740
+ Garage, Backyard, Air Conditioning, Heating, Modern Kitchen, Laundry Room, Basement, Fireplace, Swimming Pool, Home Office
741
+
742
+ Production Method
743
+ Handmade, Machine-made, Locally made, Imported, Recycled, Second-hand / Vintage, Organic, Fair trade, Custom-made, 3D printed
744
+
745
+ Pricing Basis
746
+ Fixed, Hourly, Subscription, Usage-based, Per user, Per transaction, Commission, Revenue share, Milestone-based, Retainer
747
+
748
+ Booking Channel
749
+ Direct Website, Mobile App, Online Travel Agency, Phone Call, Walk-in / In-person, Travel Agent, Global Distribution System (GDS), API / B2B Integration, Social Media Booking, Third-party Reseller
750
+
751
+ Educational Focus Area
752
+ Early Childhood Education, Primary Education, Secondary Education, Higher Education, STEM Education, Literacy, Teacher Training, Vocational Education and Training, Special Education, Adult Education
753
+
754
+ Cultural Institution Type
755
+ Museum, Public Library, Art Gallery, Theater, Cultural Center, Historic Site, Botanical Garden, Zoo, Aquarium, Science Center, Archive, Planetarium, Children's Museum, Concert Hall, Opera House, Memorial, Monument, Historic House, Visitor Center, Cultural Institute, Performing Arts Center, Community Arts Center, Living History Museum, Archaeological Site, Place of Worship, Historic District, Ethnographic Museum, Exhibition Space, Research Institute, Heritage Center, Film Archive
756
+
757
+ Medical Interventions
758
+ Medication therapy, Vaccination, Lifestyle modification, Preventive screening, Surgery, Physical therapy, Psychological therapy, Chemotherapy, Radiation therapy, Dialysis
759
+
760
+ Loyalty Program Feature
761
+ Points, Sign-up Bonus, Tiered Rewards, Cashback, Referral Program, Personalized Offers, Free Shipping, Flexible Redemption Options, Exclusive Access, Priority Customer Service
762
+
763
+ Racing Circuit Type
764
+ Permanent road course, Street circuit, Oval, Drag strip, Kart circuit, Rally stage, Hill climb, Autocross course
765
+
766
+ Competitor Brand (Fashion & Apparel)
767
+ Zara, H&M, Uniqlo, Shein, Nike, Adidas, Gap, Primark, ASOS, Mango
768
+
769
+ Hazard Type
770
+ Fire / Flammability, Slip, Trip & Fall, Electrical Hazard, Chemical Hazard (toxic, irritant, corrosive), Biological / Infectious Hazard, Mechanical Hazard (cuts, crush, entanglement), Thermal Hazard (burns, scalds, extreme cold), Choking, Noise / Hearing Damage, Ergonomic Hazard (strain, repetitive motion)
771
+
772
+ Pet Service Type
773
+ Veterinary Care, Emergency Veterinary Care, Pet Grooming, Dog Walking, Pet Sitting, Pet Boarding, Dog Daycare, Pet Training, Pet Transportation, Pet Food & Supply Delivery
774
+
775
+ Pet Wellness Package Type
776
+ Wellness/Preventive Care, Vaccination, Puppy/Kitten, Adult Pet Package, Senior, Spay/Neuter, Dental Care, Diagnostic Testing, Flea and Tick Prevention, Heartworm Prevention
777
+
778
+ Freight Transport Mode
779
+ Trucking, Ocean Freight, Rail Freight, Air Freight, Pipeline Transport, Intermodal Transport, Inland Waterway (Barge), Courier / Parcel, Last-mile Delivery
780
+
781
+ Personal Data Type
782
+ Full Name, Email Address, Phone Number, Home Address, Date of Birth, National Identification Number, Passport Number, Driver's License Number, Credit Card Number, Bank Account Number
783
+
784
+ Financial Transaction Type
785
+ Payment, Transfer, Deposit, Withdrawal, Purchase, Bill Payment, Refund, Fee, Chargeback, Currency Exchange
786
+
787
+ Spending Occasion
788
+ Groceries, Bills & utilities, Rent / Mortgage, Transportation / Commuting, Dining out / Takeout, Healthcare & medical expenses, Education / Tuition payments, Clothing, Vacation travel, Gifts
789
+
790
+ Home Improvement Project Type
791
+ Interior Painting, Exterior Painting, Kitchen Remodel, Bathroom Remodel, Flooring Installation, Roofing, Window Replacement, Door Replacement, Plumbing Repair, Electrical Upgrade, HVAC Upgrade, Siding Replacement, Insulation, Basement Finishing, Deck Construction, Fence Installation, Garage Door Replacement, Driveway Paving, Gutter Installation, Landscaping, Mold Remediation, Waterproofing, Solar Panel Installation, Home Addition, Smart Home Installation, Appliance Installation, Accessibility Modifications, Chimney Repair, Sewer & Septic Repair, Attic Conversion, Pool Installation/Repair
792
+
793
+ Ticket Category
794
+ General Admission, Reserved Seating, Standing, VIP, Premium, Early Bird, Student, Child, Senior, Accessible
795
+
796
+ Hardware Component Type
797
+ Resistor, Capacitor, Integrated Circuit, Transistor, Diode, Connector, PCB, Cable, Battery, Power Supply, Processor (CPU), Microcontroller, Memory (RAM), Storage, Display, Sensor, Actuator, Motor, Switch, LED, Oscillator, Heatsink, Fan, Antenna, Camera Module, Speaker, Microphone, Transformer, Relay, FPGA, ASIC, Power Management IC, Enclosure, Controller, Lens
798
+
799
+ Manufacturing Machine Type
800
+ CNC Machine, Lathe, Milling Machine, Drill Press, Welding Machine, Injection Molding Machine, Laser Cutter, Industrial 3D Printer, Press Brake, Grinder
801
+
802
+ Motorcycle Type
803
+ Scooter, Standard (Naked), Cruiser, Sportbike, Touring, Adventure (ADV), Dual-sport, Dirt Bike (Off-road), Electric Motorcycle, Moped
804
+
805
+ Light Color (illumination)
806
+ Warm White, Cool White, Daylight, Red, Green, Blue, Yellow, Purple, Pink, Multicolor
807
+
808
+ Endorsement Category (Products & Services)
809
+ Apparel, Beverages, Footwear, Consumer Electronics, Automotive, Beauty & Personal Care, Watches & Jewelry, Food & Snacks, Sports Equipment, Financial Services, Health & Wellness (Supplements & Fitness), Eyewear, Accessories (Bags, Hats, etc.), Home Goods & Appliances, Travel & Hospitality, Telecommunications & Internet Services, Media & Streaming Services, Gaming & Esports, Alcoholic Beverages, Fragrance & Perfume, Children's & Baby Products, Household & Cleaning Products, Pet Products, Education & Online Learning, Real Estate & Property Services, Non-profit & Cause Campaigns, Sports Teams & Leagues, Luxury Goods, Professional Services, Government & Public Service Campaigns
810
+
811
+ Communication Purpose
812
+ Informational, Notification, Reminder, Transactional, Confirmation, Promotional, Personal Connection, Support, Request, Call to Action, Alert, Welcome / Onboarding, Update / Newsletter, Invitation, Follow-up, Feedback / Survey, Instruction / How-to, Announcement, Status Update, Persuasion / Advocacy, Fundraising, Civic Duty, Candidate Support, Community Engagement, Appreciation / Thank You, Apology, Complaint, Recruitment / Hiring, Farewell / Offboarding
813
+
814
+ Motor Vehicle Collision Type
815
+ Rear-end collision, Sideswipe collision, Side-impact (T-bone) collision, Backing/reverse collision, Head-on collision, Run-off-road / single-vehicle collision, Rollover collision, Collision with fixed object, Intersection collision, Parked-vehicle / parking-lot collision, Chain-reaction / multi-vehicle collision, Pedestrian collision, Bicycle (pedalcyclist) collision, Hit-and-run collision, Animal strike (wildlife/large animal) collision, Underride / override (truck underride) collision, Collision with guardrail or barrier, Truck- or bus-involved collision, Collision with road debris, Low-speed / minor parking impact, Mechanical-failure-related collision, Other / unspecified collision type
816
+
817
+ Cultural Offering Type
818
+ Streaming Music, Streaming Films, Cinemas (Theatrical Films), Concerts, Museums, Art Galleries, Festivals, Libraries, Theater (Performing Arts), Exhibitions, Heritage Sites, Cultural Centers, Dance Performances, Opera, Ballet, Film Festivals, Comedy Shows, Literary Events, Workshops and Classes, Street Art and Public Art, Cultural Tours, Monuments and Memorials, Religious Sites and Pilgrimages, Craft Fairs and Markets, Archives and Special Collections, Virtual Exhibitions and Online Cultural Programs, Community Arts Programs, Public Lectures, Folk Events and Traditional Celebrations
819
+
820
+ Donor Classification
821
+ Individual Donor, Corporate Donor, Foundation Donor, Institutional Donor, Major Donor, Recurring Donor, First-time Donor, Lapsed Donor, Anonymous Donor, In-Kind Donor
822
+
823
+ Body Shape (Figure Type)
824
+ Rectangle, Pear, Hourglass, Apple, Inverted Triangle, Athletic, Diamond, Spoon, Oval
825
+
826
+ Personal Protective Equipment (PPE) Type
827
+ Gloves, Safety glasses/goggles, Respirator, Hard hat, Hearing protection, High-visibility clothing, Safety boots, Face shield, Safety harness, Protective clothing (coveralls/lab coat)
828
+
829
+ Academic Subject Category
830
+ Mathematics, English Language Arts, Science, Social Studies/History, Foreign Languages, Computer Science, Arts, Physical Education, Economics/Business, Engineering
831
+
832
+ Microphone Type
833
+ Dynamic, Condenser, Ribbon, Lavalier, Shotgun, USB microphone, Wireless microphone, Headset microphone, Boundary (PZM), Handheld
834
+
835
+ Pollinator Types
836
+ Bees, Butterflies, Moths, Flies, Beetles, Wasps, Ants, Birds (nectar-feeding), Bats (nectar-feeding)
837
+
838
+ Computer Peripherals
839
+ Monitor, Keyboard, Mouse, Printer, Speakers/Headphones, Webcam, Microphone, External storage (HDD/SSD/USB drive), Docking station/USB hub, Scanner
840
+
841
+ Agricultural Activities
842
+ Planting/Seeding, Irrigation, Fertilization, Pest management, Weeding, Tillage/Land preparation, Harvesting, Crop rotation, Livestock feeding/grazing, Soil testing
843
+
844
+ Mission Phase
845
+ Mission Planning, Pre-flight, Launch/Ascent, In-flight/Cruise, On-orbit/Operations, Docking/Undocking, Re-entry, Landing, Recovery/Post-flight, Abort/Contingency
846
+
847
+ Certification Level
848
+ Basic, Intermediate, Advanced, Professional, Expert, Associate, Master, Certified, Accredited, Gold
849
+
850
+ Leisure Amenities
851
+ Swimming Pool, Fitness Center, Restaurant, Bar/Lounge, Spa, Sauna/Steam Room/Hot Tub, Beach Access, Golf Course, Kids Club/Playground, Ski Facilities
852
+
853
+ Personal Relationship Type
854
+ Family, Friend, Spouse/Partner, Parent, Child, Sibling, Colleague, Neighbor, Acquaintance, Guardian
855
+
856
+ Outdoor Lighting Fixture Type
857
+ Flood light, Wall pack, Post-top, Bollard, Path light, Canopy light, Accent / spot light, Sconce, Pendant, Pole-mounted area light
858
+
859
+ Material Sourcing Type
860
+ Cotton, Organic Cotton, Recycled Cotton, Polyester, Recycled Polyester, Nylon, Recycled Nylon, Wool, Leather, Linen (Flax)
861
+
862
+ Driving Scenario
863
+ Urban Driving, Highway Driving, Parking, Lane Change, Merging, Signalized Intersection, Pedestrian Crossing, Night Driving, Rain / Wet Road Conditions, Snow / Icy Conditions
864
+
865
+ Skilled Trade
866
+ Electrician, Plumber, Carpenter, HVAC Technician, Welder, Automotive Technician, Painter, Roofer, Flooring Installer, Mason, Glazier, Drywall Installer, Tile Setter, Cabinetmaker, Locksmith, Landscaper, Heavy Equipment Operator, Concrete Finisher, Sheet Metal Worker, Pipefitter, Boilermaker, CNC Machinist, Millwright, Insulation Installer, Crane Operator, Upholsterer, Barber, Chef, Diesel Mechanic, Appliance Repair Technician, Pest Control Technician
867
+
868
+ Land Use / Development Type
869
+ Residential, Commercial, Industrial, Mixed-use, Agricultural / Farming, Parks and Recreation, Transportation / Transit Infrastructure, Logistics / Warehouse / Distribution, Institutional, Open Space / Conservation
870
+
871
+ Ritual Type
872
+ Prayer, Holiday Observance, Marriage Ceremony, Funeral, Burial, Cremation, Coming-of-Age Ceremony, Pilgrimage, Meditation, Memorial Service
873
+
874
+ High-Risk Patient Groups
875
+ Older adults (65+), People with diabetes, People with cardiovascular disease, People with chronic respiratory disease, Immunocompromised individuals, Pregnant women, People with active cancer, People with chronic kidney disease, Infants (under 1 year), Residents of long-term care facilities
876
+
877
+ Subject Area (Topic)
878
+ Politics & Government, Technology, Health, Business, Science, Education, Entertainment, Sports, Environment & Climate, Finance, Economy, Arts & Culture, Travel, History, Law & Public Policy, Psychology, Literature, Religion, Engineering, Mathematics, Food & Nutrition, Media & Journalism, Energy, Biotechnology, Sustainability, Agriculture, Human Rights, Urban Planning, Transportation, Immigration, Relationships & Family
879
+
880
+ Exhibit Theme
881
+ Natural History, Modern Art, Space Exploration, Ancient Egypt, Dinosaurs, Renaissance Art, Marine Life, Ancient Civilizations, Photography, Science and Technology, Contemporary Art, Medieval Europe, Roman Empire, Indigenous Cultures, World War II, Archaeology, Industrial Revolution, Fashion and Costume, Local History, Interactive Science, Human Evolution, Botany, Environmental Conservation, Maritime History, Textiles, Children's Exhibits, Music and Performing Arts, Sports History, Early Settlement, Design and Architecture, Medical History, Cultural Heritage
882
+
883
+ Venue Type
884
+ Bar, Restaurant, Club, Theater, Cinema, Concert Hall, Auditorium, Stadium, Arena, Convention/Conference Center, Amphitheater, Ballroom, Banquet Hall, Museum, Gallery, Casino, Sports Field, Sports Complex, Community Center, Park, Plaza, Rooftop Venue, Campus Venue, House of Worship, Private Residence, Comedy Club, Bowling Alley, Boat/Ship, Library, Racecourse
885
+
886
+ Toy Type
887
+ Construction & Building Sets, Dolls, Plush Toys (Stuffed Animals), Action Figures, Puzzles, Board Games, Toy Vehicles, Educational & STEM Toys, Pretend Play & Dress-up, Arts & Crafts Kits, Electronic & Interactive Toys, Remote Control Vehicles, Playsets (dollhouses, themed sets), Infant & Toddler Toys (stacking, shape sorters, lacing), Musical Toy Instruments, Ride-on Toys, Fidget & Sensory Toys, Trains & Train Sets, Marble Runs & Ball-Drop Toys, Model Kits & Hobby Sets, Die-cast Vehicles & Collectible Cars, Outdoor & Sports Toys, Bath Toys, Puppets, Card & Collectible Games
888
+
889
+ Harvesting Method
890
+ Mechanized harvesting, Hand harvesting, Combine harvester, Forage harvester, Baling, Windrowing, Tree shaker, Selective harvesting
891
+
892
+ Hazard Mitigation Project Type
893
+ Stormwater management, Levee rehabilitation and upgrade, Seawall construction and repair, Wetland restoration and creation, Floodplain restoration, Living shorelines, Beach nourishment, Reservoir construction, Storm surge barriers and gates, Pumping stations (stormwater/sea), Retention and detention basins, Dredging and sediment management, River channel restoration, Diversion channel construction, Culvert and bridge modifications, Bank stabilization, Coastal barrier enhancement, Breakwater construction, Groin construction, Dam removal (river restoration), Permeable pavement, Bioswales and rain gardens, Green roofs, Reforestation and afforestation, Slope stabilization (retaining walls, terraces), Floodproofing and structure elevation, Property acquisition and buyouts (managed retreat), Land-use planning and zoning measures, Early warning and monitoring systems, Evacuation route and emergency access improvements, Wildfire fuel reduction and prescribed burning, Firebreaks and defensible space
894
+
895
+ Accessibility Feature Type
896
+ Elevators, Ramps, Accessible parking spaces, Automatic doors, Wide doorways and corridors, Accessible restrooms, Curb ramps (curb cuts), Grab bars, Step-free / accessible routes, Braille signage, Tactile paving, Visual alarms (strobe alerts), Audio announcements, Hearing loop / induction loop systems, Lever door handles, Rocker light switches, Lowered countertops and service counters, Roll-in showers and accessible bathing, Vertical platform lifts and stairlifts, Accessible seating (priority and companion seating), High-contrast signage and markings, Accessible website / digital accessibility (WCAG), Captions and subtitles (video accessibility), Sign language interpretation services, Wayfinding and orientation features (including tactile maps)
897
+
898
+ Tractor Powertrain & Control Configuration
899
+ Diesel, Battery-electric, 2WD (two-wheel drive), 4WD (four-wheel drive), Tracked (crawler), Manual (operator-controlled), Auto-steer / GPS-assisted (semi-autonomous guidance), Autonomous (fully autonomous)
900
+
901
+ Filtration Technology
902
+ Activated Carbon Adsorption, Sand Filtration, Cartridge Filter, HEPA Filter, Reverse Osmosis, Microfiltration, Ultrafiltration, Nanofiltration, Ion Exchange, Filter Press
903
+
904
+ Running Shoe Segment
905
+ Neutral / Cushioned, Stability, Daily Trainer, Racing / Racing Flat, Carbon-Plated Racing Shoe, Trail Running, Minimalist / Barefoot, Maximalist / Max Cushion, Kids Running Shoe
906
+
907
+ Gift Type
908
+ Cash, Gift Card, Physical Gift, Digital Gift, Experience Gift, Subscription Gift, Flowers, Electronics, Clothing, Food & Beverage
909
+
910
+ Farm Diversification Strategies
911
+ Crop rotation, Crop diversification, Livestock integration, Direct-to-consumer sales, Value-added processing, Agritourism, Organic certification, Agroforestry, Renewable energy production, Community-supported agriculture (CSA)
912
+
913
+ Payment Method
914
+ Cash, Credit Card, Debit Card, Bank Transfer, Digital Wallet, PayPal, Check, Gift Card, Buy Now Pay Later, Cryptocurrency
915
+
916
+ Climate Zone (major types — Köppen & common names)
917
+ Tropical Rainforest, Tropical Savanna, Desert, Semi-arid (Steppe), Humid Subtropical, Oceanic (Marine West Coast), Mediterranean, Humid Continental, Tropical Monsoon, Temperate (general), Subarctic (Boreal), Tundra, Alpine (Highland), Polar (Arctic/Antarctic), Ice Cap
918
+
919
+ Architectural Style
920
+ Contemporary, Victorian, Craftsman, Colonial, Ranch, Mid-Century Modern, Farmhouse, Bungalow, Tudor, Mediterranean Revival, Spanish Colonial, Cape Cod, Art Deco, Industrial, Split-level, Neoclassical, Dutch Colonial, Log Cabin, Georgian, International Style, Federal, Mission Revival, Beaux-Arts, Prairie, Gothic Revival, Art Nouveau, Brutalist, Renaissance Revival
921
+
922
+ Lighting Fixture Type
923
+ Recessed Lighting, Ceiling Light, Flush Mount Ceiling Light, Pendant Light, Ceiling Fan, Chandelier, Wall Sconce, Table Lamp, Floor Lamp, Track Lighting, Vanity Light, Under-Cabinet Lighting, Linear Suspension, Semi-Flush Mount Ceiling Light, Spotlight, Floodlight, LED Strip Lighting, Accent Lighting, Picture Light, Landscape Lighting, Cove Lighting, Rope Light, Lantern (Outdoor Lantern), Path Light, Post Light, Solar Light, Bollard Light, Step Light, High Bay Light, Utility/Shop Light
924
+
925
+ Washing Machine Type
926
+ Front-load, Top-load agitator, Top-load impeller, Washer-dryer combo (all-in-one), Stackable washer-dryer (separate stacked units), Portable washing machine, Compact / undercounter washer, Semi-automatic twin-tub, Laundry center (single-unit stacked washer-dryer), Coin-operated / commercial washer, Industrial / high-capacity washer, Drawer washing machine
927
+
928
+ Art Supply Set Type
929
+ Colored Pencil Set, Watercolor Set, Acrylic Paint Set, Marker Set, Crayon Set, Graphite Pencil Set, Sketching Set, Charcoal Set, Oil Paint Set, Mixed Media Set
930
+
931
+ Art & Craft Workshop Discipline
932
+ Painting, Drawing, Photography, Sculpture, Ceramics, Textiles, Printmaking, Jewelry Making, Woodworking, Digital Art
933
+
934
+ Scientific Discovery Category
935
+ New Species, New Technology, New Medical Treatment, New Disease or Pathogen, Archaeological Site, Archaeological Artifact, New Fossil, Exoplanet, New Material, New Scientific Method or Technique
936
+
937
+ Monetization Method
938
+ Advertising, Subscriptions, Physical Product Sales, One-time Purchase, In-App Purchases, Freemium, Transaction Fees, Affiliate Marketing, Licensing and Royalties, Donations and Crowdfunding
939
+
940
+ Debris Material
941
+ Wood, Concrete, Metal, Plastic, Glass, Paper / Cardboard, Soil / Dirt, Asphalt, Electronic waste, Vegetation (green waste)
942
+
943
+ Fishing Capture Method
944
+ Trawl, Purse seine, Gillnet, Longline, Pots and traps, Trolling, Handline, Pole-and-line, Dredge, Spearfishing
945
+
946
+ Land Management Practice
947
+ Crop rotation, Cover cropping, Integrated pest management, Irrigation management, Conservation tillage, Reforestation, Agroforestry, Rotational grazing, Wetland restoration, Prescribed burning
948
+
949
+ Basketball Shot Zone
950
+ Restricted Area, In the Paint (Non-RA), Above the Break 3, Top of Key 3, Left Wing 3, Right Wing 3, Left Corner 3, Right Corner 3, Mid-Range, Elbow (Mid-Range), Long Two, Backcourt / Heave
951
+
952
+ Dress Silhouette
953
+ A-line, Sheath, Fit-and-flare, Wrap dress, Shift, Empire waist, Ball gown, Mermaid, Bodycon, Slip dress
954
+
955
+ Water Conservation Measure
956
+ Water-efficient fixtures, Leak detection and repair, Smart water meters, Drought-tolerant landscaping, Rainwater harvesting, Drip irrigation, Greywater reuse, Water audits, Rebate and incentive programs, Water use restrictions and watering schedules
957
+
icon_generation/image_gen.py ADDED
@@ -0,0 +1,119 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ import json
3
+ from openai import OpenAI
4
+ from PIL import Image
5
+ from io import BytesIO
6
+ import base64
7
+ import sys
8
+ import time
9
+ import multiprocessing
10
+ from concurrent.futures import ProcessPoolExecutor
11
+
12
+ "The output must be a single image with an exact 6:4 aspect ratio (landscape orientation, width greater than height). The image must contain exactly 24 icons arranged in a strict grid of 6 columns (horizontal, left to right) and 4 rows (vertical, top to bottom). Do not rotate, transpose, or alter the grid orientation. No text, letters, numbers, labels, or captions. Use a pure white background only."
13
+
14
+ API_KEY = os.getenv("OPENAI_API_KEY") or os.getenv("AIHUBMIX_API_KEY", "")
15
+ API_PROVIDER = os.getenv("OPENAI_BASE_URL", "https://aihubmix.com/v1")
16
+
17
+ # 创建OpenAI客户端的函数,每个进程需要自己的客户端实例
18
+ def create_client():
19
+ return OpenAI(
20
+ api_key=API_KEY,
21
+ base_url=API_PROVIDER,
22
+ )
23
+
24
+ # 全局客户端仅用于主进程
25
+ client = create_client()
26
+
27
+ def generate_image(task):
28
+ """生成指定attribute和value的图像"""
29
+ attribute, value = task
30
+
31
+ # 每个进程创建自己的客户端实例
32
+ local_client = create_client()
33
+
34
+ prompt = f"""Create a flat-design colorful pictogram symbolizing '{attribute}: {value}'. The design should be simple, no text or intricate details, no shading or gradients, and set against a white background."""
35
+
36
+ # 检查图像是否已存在
37
+ output_path = os.path.join("images", f"{attribute}-{value}.png")
38
+ if os.path.exists(output_path):
39
+ print(f"图像已存在,跳过生成: {output_path}")
40
+ return None
41
+
42
+ try:
43
+ result = local_client.images.generate(
44
+ model="gpt-image-1",
45
+ prompt=prompt,
46
+ n=1,
47
+ size="1024x1024",
48
+ quality="low",
49
+ moderation="low",
50
+ background="auto",
51
+ )
52
+
53
+ print(f"生成 {attribute}: {value} 的图像成功")
54
+
55
+ # 立即保存图像
56
+ if result and result.data:
57
+ image_base64 = result.data[0].b64_json
58
+ if image_base64:
59
+ image_bytes = base64.b64decode(image_base64)
60
+ with open(output_path, "wb") as f:
61
+ f.write(image_bytes)
62
+ print(f"图片已保存至:{output_path}")
63
+ return True
64
+
65
+ return None
66
+ except Exception as e:
67
+ print(f"生成 {attribute}: {value} 的图像失败: {e}")
68
+ return None
69
+
70
+ def main():
71
+ # 读取filtered.json文件
72
+ try:
73
+ with open("filtered.json", "r", encoding="utf-8") as f:
74
+ data = json.load(f)
75
+ except Exception as e:
76
+ print(f"读取filtered.json失败: {e}")
77
+ return
78
+
79
+ # 统计attribute-value对的总数
80
+ total_pairs = 0
81
+ attribute_counts = {}
82
+ for item in data:
83
+ attribute = item["name"]
84
+ value_count = len(item["values"])
85
+ total_pairs += value_count
86
+ attribute_counts[attribute] = value_count
87
+
88
+ print(f"总共发现 {total_pairs} 个attribute-value对")
89
+
90
+ # 确认是否继续
91
+ user_input = input("确认开始生成图像? (y/n): ")
92
+ if user_input.lower() != 'y':
93
+ print("已取消生成")
94
+ return
95
+
96
+ # 创建输出目录
97
+ os.makedirs("images", exist_ok=True)
98
+
99
+ # 创建任务列表
100
+ tasks = []
101
+ for item in data:
102
+ attribute = item["name"]
103
+ for value_info in item["values"]:
104
+ value = value_info["value"]
105
+ tasks.append((attribute, value))
106
+
107
+ # 设置进程数量,根据CPU核心数量确定
108
+ num_processes = min(8, multiprocessing.cpu_count())
109
+ print(f"使用 {num_processes} 个进程并行处理")
110
+
111
+ # 使用进程池并行处理
112
+ with ProcessPoolExecutor(max_workers=num_processes) as executor:
113
+ executor.map(generate_image, tasks)
114
+
115
+ print("所有图像生成任务已完成")
116
+
117
+ if __name__ == "__main__":
118
+ multiprocessing.freeze_support() # Windows系统需要
119
+ main()
icon_generation/refine_domains.py ADDED
@@ -0,0 +1,537 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """
3
+ 使用 LLM 对 filtered.json 中的 name 字段进行 refine,
4
+ 将其改为更明确的 domain,并按 count 排序
5
+ """
6
+
7
+ import json
8
+ import os
9
+ import requests
10
+ from typing import Dict, Optional, List
11
+ from collections import Counter
12
+ from concurrent.futures import ThreadPoolExecutor, as_completed
13
+ import threading
14
+
15
+
16
+ class DomainRefiner:
17
+ """使用 LLM 来 refine domain names"""
18
+
19
+ def __init__(self, api_key=None, base_url=None, model=None):
20
+ """
21
+ 初始化 LLM analyzer
22
+
23
+ Args:
24
+ api_key: API key
25
+ base_url: API base URL
26
+ model: Model name
27
+ """
28
+ self.api_key = api_key or os.getenv("OPENAI_API_KEY") or os.getenv("AIHUBMIX_API_KEY", "")
29
+ self.base_url = base_url or os.getenv("OPENAI_BASE_URL", "https://aihubmix.com/v1")
30
+ self.model = model or os.getenv("OPENAI_MODEL", "gemini-2.5-flash")
31
+ self.lock = threading.Lock() # 线程锁,用于打印
32
+
33
+ def filter_values(self, domain_name: str, values_with_counts: List[Dict]) -> List[Dict]:
34
+ """
35
+ 使用 LLM 过滤掉不属于该 domain 的 specific values 和相似/重复的 values
36
+
37
+ Args:
38
+ domain_name: domain 名称
39
+ values_with_counts: 包含 value 和 count 的列表 [{"value": "...", "count": ...}, ...]
40
+
41
+ Returns:
42
+ 过滤后的 values 列表
43
+ """
44
+ # 如果值太多,只发送 top 50 给 LLM
45
+ values_to_check = values_with_counts[:50] if len(values_with_counts) > 50 else values_with_counts
46
+
47
+ prompt = self._build_filter_prompt(domain_name, values_to_check)
48
+
49
+ try:
50
+ response = self._query_llm(prompt)
51
+
52
+ if response:
53
+ # 清理可能的 markdown 代码块
54
+ cleaned_response = response.strip()
55
+ if cleaned_response.startswith('```'):
56
+ lines = cleaned_response.split('\n')
57
+ cleaned_response = '\n'.join(lines[1:-1] if lines[-1].strip() == '```' else lines[1:])
58
+ cleaned_response = cleaned_response.replace('```json', '').replace('```', '').strip()
59
+
60
+ result = json.loads(cleaned_response)
61
+
62
+ # 验证返回格式
63
+ if 'filtered_values' in result:
64
+ filtered_values_set = set(result['filtered_values'])
65
+
66
+ # 对于 top 50,过滤它们
67
+ filtered_top = [v for v in values_to_check if v['value'] in filtered_values_set]
68
+
69
+ # 如果原始列表更长,保留剩余的(因为没有检查)
70
+ if len(values_with_counts) > 50:
71
+ remaining = values_with_counts[50:]
72
+ filtered_list = filtered_top + remaining
73
+ else:
74
+ filtered_list = filtered_top
75
+
76
+ return filtered_list
77
+ else:
78
+ with self.lock:
79
+ print(f" ⚠️ LLM 过滤响应缺少字段,保留原始值")
80
+ return values_with_counts
81
+
82
+ else:
83
+ with self.lock:
84
+ print(f" ⚠️ LLM 过滤失败,保留原始值")
85
+ return values_with_counts
86
+
87
+ except json.JSONDecodeError as e:
88
+ with self.lock:
89
+ print(f" ⚠️ LLM 过滤响应不是有效的 JSON,保留原始值")
90
+ return values_with_counts
91
+ except Exception as e:
92
+ with self.lock:
93
+ print(f" ⚠️ 值过滤错误: {e},保留原始值")
94
+ return values_with_counts
95
+
96
+ def _build_filter_prompt(self, domain_name: str, values_with_counts: List[Dict]) -> str:
97
+ """构建值过滤的 prompt"""
98
+
99
+ values_str = '\n'.join([f' - "{v["value"]}" (count: {v["count"]})' for v in values_with_counts])
100
+
101
+ prompt = f"""Given a domain name and its associated values, please filter out:
102
+ 1. Values that don't truly belong to this domain (too specific, off-topic, or irrelevant)
103
+ 2. Similar or duplicate values (keep the most common or representative one)
104
+ 3. Values that are too generic or ambiguous
105
+
106
+ Domain: "{domain_name}"
107
+
108
+ Values to filter:
109
+ {values_str}
110
+
111
+ Please analyze these values and return ONLY the values that:
112
+ - Clearly belong to this domain
113
+ - Are distinct (not duplicates or very similar)
114
+ - Are meaningful attributes
115
+
116
+ Return your response in the following JSON format ONLY (no markdown, no extra text):
117
+ {{
118
+ "filtered_values": ["value1", "value2", ...],
119
+ "removed_count": number_of_removed_values,
120
+ "reasoning": "Brief explanation of filtering criteria used"
121
+ }}
122
+
123
+ Examples:
124
+ - Domain "Movie Genre" with values ["Action", "action movie", "ACT"] → Keep only "Action"
125
+ - Domain "Country" with values ["USA", "New York", "California"] → Remove "New York", "California" (cities, not countries)
126
+ """
127
+
128
+ return prompt
129
+
130
+ def refine_domain_name(self, name: str, sample_values: List[str], total_count: int) -> Dict:
131
+ """
132
+ 使用 LLM 分析并 refine domain name
133
+
134
+ Args:
135
+ name: 原始 name
136
+ sample_values: 一些示例 values
137
+ total_count: 总计数
138
+
139
+ Returns:
140
+ {
141
+ 'original_name': str,
142
+ 'refined_domain': str,
143
+ 'reasoning': str
144
+ }
145
+ """
146
+ prompt = self._build_domain_refinement_prompt(name, sample_values, total_count)
147
+
148
+ try:
149
+ response = self._query_llm(prompt)
150
+
151
+ if response:
152
+ # 清理可能的 markdown 代码块
153
+ cleaned_response = response.strip()
154
+ if cleaned_response.startswith('```'):
155
+ lines = cleaned_response.split('\n')
156
+ cleaned_response = '\n'.join(lines[1:-1] if lines[-1].strip() == '```' else lines[1:])
157
+ cleaned_response = cleaned_response.replace('```json', '').replace('```', '').strip()
158
+
159
+ result = json.loads(cleaned_response)
160
+
161
+ # 验证返回格式
162
+ if 'refined_domain' in result:
163
+ result['original_name'] = name
164
+ return result
165
+ else:
166
+ with self.lock:
167
+ print(f" ❌ LLM 响应缺少必需字段: {result}")
168
+ return None
169
+
170
+ else:
171
+ with self.lock:
172
+ print(" ❌ LLM API 调用失败")
173
+ return None
174
+
175
+ except json.JSONDecodeError as e:
176
+ with self.lock:
177
+ print(f" ❌ LLM 响应不是有效的 JSON: {e}")
178
+ print(f" 响应内容: {response[:500]}...")
179
+ return None
180
+ except Exception as e:
181
+ with self.lock:
182
+ print(f" ❌ Domain refinement 错误: {e}")
183
+ return None
184
+
185
+ def _build_domain_refinement_prompt(self, name: str, sample_values: List[str], total_count: int) -> str:
186
+ """构建 domain refinement 的 prompt"""
187
+
188
+ sample_values_str = ', '.join(f'"{v}"' for v in sample_values[:10])
189
+
190
+ prompt = f"""Given a data field name and its sample values, please refine the name to a more specific and clear domain name.
191
+
192
+ The domain name should:
193
+ 1. Clearly indicate what category/dimension this field represents
194
+ 2. Be consistent and professional
195
+ 3. Form a clear "domain: specific attribute" relationship with its values
196
+ 4. Be concise (1-3 words)
197
+
198
+ Input Information:
199
+ - Original Field Name: "{name}"
200
+ - Sample Values: {sample_values_str}
201
+ - Total Entries Count: {total_count}
202
+
203
+ Please analyze the field name and sample values, then provide:
204
+ 1. A refined domain name that better describes this dimension
205
+ 2. Brief reasoning for your choice
206
+
207
+ Return your response in the following JSON format ONLY (no markdown, no extra text):
208
+ {{
209
+ "refined_domain": "YourRefinedDomainName",
210
+ "reasoning": "Brief explanation of why this domain name is more appropriate"
211
+ }}
212
+
213
+ Examples:
214
+ - Original: "Genre" with values ["Action", "Rock", "Pop"] → Refined: "Entertainment Genre"
215
+ - Original: "Type" with values ["Movie", "TV Show"] → Refined: "Media Type"
216
+ - Original: "Category" with values ["Electronics", "Books"] → Refined: "Product Category"
217
+ """
218
+
219
+ return prompt
220
+
221
+ def _query_llm(self, prompt: str) -> Optional[str]:
222
+ """
223
+ 查询 LLM API
224
+
225
+ Args:
226
+ prompt: 发送给 LLM 的 prompt
227
+
228
+ Returns:
229
+ str: LLM 响应内容
230
+ """
231
+ headers = {
232
+ 'Authorization': f'Bearer {self.api_key}',
233
+ 'Content-Type': 'application/json'
234
+ }
235
+
236
+ data = {
237
+ 'model': self.model,
238
+ 'messages': [
239
+ {
240
+ 'role': 'system',
241
+ 'content': 'You are a data modeling expert specialized in creating clear, semantic domain names. Always return valid JSON format only, without any markdown formatting or extra text.'
242
+ },
243
+ {
244
+ 'role': 'user',
245
+ 'content': prompt
246
+ }
247
+ ],
248
+ 'temperature': 0.3
249
+ }
250
+
251
+ try:
252
+ response = requests.post(
253
+ f'{self.base_url}/chat/completions',
254
+ headers=headers,
255
+ json=data,
256
+ timeout=30
257
+ )
258
+ response.raise_for_status()
259
+
260
+ result = response.json()
261
+ return result['choices'][0]['message']['content'].strip()
262
+
263
+ except requests.exceptions.Timeout:
264
+ with self.lock:
265
+ print("❌ LLM API 超时")
266
+ return None
267
+ except requests.exceptions.HTTPError as e:
268
+ with self.lock:
269
+ print(f"❌ LLM API HTTP 错误: {e}")
270
+ if hasattr(e.response, 'text'):
271
+ print(f" 响应: {e.response.text[:200]}")
272
+ return None
273
+ except requests.exceptions.RequestException as e:
274
+ with self.lock:
275
+ print(f"❌ LLM API 请求错误: {e}")
276
+ return None
277
+ except KeyError as e:
278
+ with self.lock:
279
+ print(f"❌ LLM API 响应格式错误: {e}")
280
+ return None
281
+
282
+
283
+ def process_single_item(item: Dict, idx: int, total: int, refiner: DomainRefiner, start_time: float, start_idx: int) -> Dict:
284
+ """
285
+ 处理单个数据项
286
+
287
+ Args:
288
+ item: 数据项
289
+ idx: 当前索引
290
+ total: 总数量
291
+ refiner: DomainRefiner 实例
292
+ start_time: 开始时间
293
+ start_idx: 起始索引
294
+
295
+ Returns:
296
+ 处理后的数据项
297
+ """
298
+ import time
299
+
300
+ # 计算进度信息
301
+ progress = (idx + 1) / total * 100
302
+ elapsed = time.time() - start_time
303
+ avg_time = elapsed / (idx - start_idx + 1) if idx > start_idx else 0
304
+ remaining = avg_time * (total - idx - 1)
305
+
306
+ with refiner.lock:
307
+ print(f"\n{'=' * 80}")
308
+ print(f"🔄 处理进度: {idx + 1}/{total} ({progress:.1f}%)")
309
+ print(f"⏱️ 已用时间: {elapsed:.1f}秒 | 预计剩余: {remaining:.1f}秒")
310
+ print(f"📝 当前字段: {item['name']}")
311
+ print(f" - 值数量: {item['num_values']}")
312
+ print(f" - 总计数: {item['total_count']}")
313
+
314
+ # 提取 sample values
315
+ sample_values = [v['value'] for v in item['values'][:15]]
316
+
317
+ with refiner.lock:
318
+ print(f" - 示例值: {', '.join(sample_values[:5])}")
319
+ print(f"🤖 调用 LLM 进行 domain refinement...")
320
+
321
+ # 使用 LLM refine domain name
322
+ refined_result = refiner.refine_domain_name(
323
+ item['name'],
324
+ sample_values,
325
+ item['total_count']
326
+ )
327
+
328
+ if refined_result:
329
+ refined_domain = refined_result['refined_domain']
330
+ reasoning = refined_result.get('reasoning', '')
331
+
332
+ with refiner.lock:
333
+ print(f"✅ Domain refinement 成功!")
334
+ print(f" 原始名称: '{item['name']}'")
335
+ print(f" 优化域名: '{refined_domain}'")
336
+ print(f" 优化理由: {reasoning}")
337
+ print(f"🔍 调用 LLM 进行 value filtering...")
338
+
339
+ # 过滤 values
340
+ filtered_values = refiner.filter_values(refined_domain, item['values'])
341
+
342
+ with refiner.lock:
343
+ removed_count = len(item['values']) - len(filtered_values)
344
+ print(f"✅ Value filtering 完成!")
345
+ print(f" 原始值数量: {len(item['values'])}")
346
+ print(f" 过滤后数量: {len(filtered_values)}")
347
+ print(f" 移除数量: {removed_count}")
348
+
349
+ # 构建新的数据项
350
+ refined_item = {
351
+ 'domain': refined_domain,
352
+ 'original_name': item['name'],
353
+ 'num_values': len(filtered_values),
354
+ 'original_num_values': item['num_values'],
355
+ 'total_count': sum(v['count'] for v in filtered_values),
356
+ 'original_total_count': item['total_count'],
357
+ 'values': filtered_values,
358
+ 'refinement_reasoning': reasoning
359
+ }
360
+
361
+ return refined_item
362
+ else:
363
+ # 如果 LLM 失败,保留原始 name 作为 domain
364
+ with refiner.lock:
365
+ print(f"⚠️ LLM refinement 失败,使用原始 name,跳过值过滤")
366
+
367
+ refined_item = {
368
+ 'domain': item['name'],
369
+ 'original_name': item['name'],
370
+ 'num_values': item['num_values'],
371
+ 'original_num_values': item['num_values'],
372
+ 'total_count': item['total_count'],
373
+ 'original_total_count': item['total_count'],
374
+ 'values': item['values'],
375
+ 'refinement_reasoning': 'LLM refinement failed, kept original'
376
+ }
377
+
378
+ return refined_item
379
+
380
+
381
+ def process_filtered_json(input_file: str, output_file: str, temp_file: str = None, num_threads: int = 10):
382
+ """
383
+ 处理 filtered.json 文件(并行处理)
384
+
385
+ Args:
386
+ input_file: 输入文件路径
387
+ output_file: 输出文件路径
388
+ temp_file: 临时文件路径,用于实时保存中间结果
389
+ num_threads: 并行线程数
390
+ """
391
+ import os
392
+ import time
393
+
394
+ if temp_file is None:
395
+ temp_file = output_file.replace('.json', '_temp.json')
396
+
397
+ print(f"📖 读取文件: {input_file}")
398
+ print(f"💾 临时文件: {temp_file}")
399
+ print(f"✨ 最终文件: {output_file}")
400
+ print(f"🔧 并行线程数: {num_threads}")
401
+
402
+ # 读取原始数据
403
+ with open(input_file, 'r', encoding='utf-8') as f:
404
+ data = json.load(f)
405
+
406
+ print(f"✅ 成功读取 {len(data)} 个字段\n")
407
+ print("=" * 80)
408
+
409
+ # 检查是否已有临时文件(支持断点续传)
410
+ refined_data = []
411
+ start_idx = 0
412
+ processed_names = set()
413
+
414
+ if os.path.exists(temp_file):
415
+ print(f"📂 发现临时文件,尝试恢复进度...")
416
+ try:
417
+ with open(temp_file, 'r', encoding='utf-8') as f:
418
+ refined_data = json.load(f)
419
+ start_idx = len(refined_data)
420
+ processed_names = {item['original_name'] for item in refined_data}
421
+ print(f"✅ 已恢复 {start_idx} 个字段的处理结果")
422
+ except Exception as e:
423
+ print(f"⚠️ 临时文件读取失败: {e},从头开始")
424
+ refined_data = []
425
+ start_idx = 0
426
+ processed_names = set()
427
+
428
+ # 过滤掉已处理的项
429
+ items_to_process = [item for item in data if item['name'] not in processed_names]
430
+
431
+ if not items_to_process:
432
+ print("✅ 所有项目已处理完成!")
433
+ return
434
+
435
+ print(f"📋 待处理项目: {len(items_to_process)} 个")
436
+ print("=" * 80)
437
+
438
+ # 初始化 LLM refiner
439
+ refiner = DomainRefiner()
440
+
441
+ # 记录开始时间
442
+ start_time = time.time()
443
+
444
+ # 使用线程池并行处理
445
+ with ThreadPoolExecutor(max_workers=num_threads) as executor:
446
+ # 提交所有任务
447
+ future_to_idx = {
448
+ executor.submit(
449
+ process_single_item,
450
+ item,
451
+ start_idx + i,
452
+ len(data),
453
+ refiner,
454
+ start_time,
455
+ start_idx
456
+ ): (start_idx + i, item)
457
+ for i, item in enumerate(items_to_process)
458
+ }
459
+
460
+ # 按完成顺序收集结果
461
+ completed = 0
462
+ for future in as_completed(future_to_idx):
463
+ idx, item = future_to_idx[future]
464
+ try:
465
+ result = future.result()
466
+ refined_data.append(result)
467
+ completed += 1
468
+
469
+ # 每处理 5 个项目保存一次
470
+ if completed % 5 == 0:
471
+ with refiner.lock:
472
+ print(f"\n{'=' * 80}")
473
+ print(f"💾 保存中间结果... (已完成 {completed}/{len(items_to_process)})")
474
+ with open(temp_file, 'w', encoding='utf-8') as f:
475
+ json.dump(refined_data, f, ensure_ascii=False, indent=2)
476
+ with refiner.lock:
477
+ print(f"✅ 已保存 {len(refined_data)} 个字段到临时文件")
478
+
479
+ except Exception as e:
480
+ with refiner.lock:
481
+ print(f"\n❌ 处理项目 {idx} ({item['name']}) 时出错: {e}")
482
+
483
+ # 最终保存一次
484
+ print(f"\n{'=' * 80}")
485
+ print(f"💾 保存最终临时结果...")
486
+ with open(temp_file, 'w', encoding='utf-8') as f:
487
+ json.dump(refined_data, f, ensure_ascii=False, indent=2)
488
+
489
+ # 按 total_count 降序排序
490
+ print(f"\n{'=' * 80}")
491
+ print("📊 按 total_count 进行排序...")
492
+ refined_data.sort(key=lambda x: x['total_count'], reverse=True)
493
+
494
+ # 保存最终结果
495
+ print(f"💾 保存最终结果到: {output_file}")
496
+ with open(output_file, 'w', encoding='utf-8') as f:
497
+ json.dump(refined_data, f, ensure_ascii=False, indent=2)
498
+
499
+ print(f"\n{'=' * 80}")
500
+ print(f"✅ 处理完成!共处理 {len(refined_data)} 个字段")
501
+ print(f"⏱️ 总耗时: {time.time() - start_time:.1f}秒")
502
+
503
+ # 输出统计信息
504
+ print(f"\n{'=' * 80}")
505
+ print("📈 统计信息:")
506
+ print(f" 总字段数: {len(refined_data)}")
507
+ print(f" 总记录数(过滤后): {sum(item['total_count'] for item in refined_data):,}")
508
+ print(f" 总记录数(原始): {sum(item.get('original_total_count', item['total_count']) for item in refined_data):,}")
509
+
510
+ # 计算过滤统计
511
+ total_values_before = sum(item.get('original_num_values', item['num_values']) for item in refined_data)
512
+ total_values_after = sum(item['num_values'] for item in refined_data)
513
+ print(f" 总值数量(原始): {total_values_before:,}")
514
+ print(f" 总值数量(过滤后): {total_values_after:,}")
515
+ print(f" 过滤比例: {(1 - total_values_after/total_values_before)*100:.1f}%")
516
+
517
+ # 显示前 10 个 domain
518
+ print(f"\n{'=' * 80}")
519
+ print("🏆 Top 10 Domains (按 total_count):")
520
+ for i, item in enumerate(refined_data[:10]):
521
+ print(f"\n {i+1}. {item['domain']} (原: {item['original_name']})")
522
+ print(f" Count: {item['total_count']:,}, Values: {item['num_values']}")
523
+ print(f" 示例值: {', '.join([v['value'] for v in item['values'][:5]])}")
524
+
525
+ # 删除临时文件
526
+ if os.path.exists(temp_file):
527
+ print(f"\n🗑�� 保留临时文件以备恢复: {temp_file}")
528
+ # os.remove(temp_file) # 暂时不删除,以便需要时恢复
529
+
530
+
531
+ if __name__ == '__main__':
532
+ input_file = '/home/lizhen/ChartPipeline/icon_generation/filtered.json'
533
+ output_file = '/home/lizhen/ChartPipeline/icon_generation/refined_domains.json'
534
+
535
+ process_filtered_json(input_file, output_file, num_threads=10)
536
+
537
+
icon_generation/split_icon.py ADDED
@@ -0,0 +1,487 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import cv2
2
+ import numpy as np
3
+ import os
4
+ import re
5
+ import glob
6
+
7
+ def clean_filename(text):
8
+ """清理文件名"""
9
+ text = text.lower().strip()
10
+ text = re.sub(r'[\s\W]+', '_', text)
11
+ return text.strip('_')
12
+
13
+ def sort_contours(cnts, method="left-to-right"):
14
+ """
15
+ 对轮廓进行排序。
16
+ 对于 Grid 布局,我们需要 'top-to-bottom' 然后 'left-to-right' 的混合排序。
17
+ """
18
+ if not cnts:
19
+ return [], []
20
+
21
+ # 获取每个轮廓的 Bounding Box
22
+ boundingBoxes = [cv2.boundingRect(c) for c in cnts]
23
+
24
+ # 将轮廓和bbox打包
25
+ cnts_boxes = list(zip(cnts, boundingBoxes))
26
+
27
+ # 1. 按照 Y 坐标(从上到下)进行初步排序
28
+ # key: y
29
+ cnts_boxes.sort(key=lambda b: b[1][1])
30
+
31
+ # 2. 分行处理
32
+ # 由于手工画线或扫描误差,同一行的y坐标可能不完全相同。
33
+ # 我们需要设定一个阈值,认为y坐标相近的是“同一行”。
34
+ rows = []
35
+ current_row = []
36
+ if cnts_boxes:
37
+ # 以第一个轮廓的高度作为参考阈值
38
+ ref_h = cnts_boxes[0][1][3]
39
+ tolerance = ref_h * 0.5 # 容差设为高度的一半
40
+
41
+ last_y = cnts_boxes[0][1][1]
42
+
43
+ for c, box in cnts_boxes:
44
+ y = box[1]
45
+ if y <= last_y + tolerance:
46
+ current_row.append((c, box))
47
+ else:
48
+ # 新的一行
49
+ rows.append(current_row)
50
+ current_row = [(c, box)]
51
+ last_y = y
52
+ # 添加最后一行
53
+ if current_row:
54
+ rows.append(current_row)
55
+
56
+ # 3. 对每一行内部,按照 X 坐标(从左到右)排序
57
+ final_sorted = []
58
+ row_counts = []
59
+ for i, row in enumerate(rows):
60
+ # key: x
61
+ row.sort(key=lambda b: b[1][0])
62
+ row_counts.append(len(row))
63
+ for item in row:
64
+ final_sorted.append(item[1]) # 只返回 bbox (x, y, w, h)
65
+
66
+ return final_sorted, row_counts
67
+
68
+ def uniform_grid_split(img, expected_cols=6, expected_rows=4, margin_percent=0.02):
69
+ """
70
+ 均匀分割方法:直接按照预期的行列数均匀分割图像
71
+ 适用于网格线不连续或没有明显网格线的情况
72
+
73
+ Args:
74
+ img: 输入图像
75
+ expected_cols: 期望的列数
76
+ expected_rows: 期望的行数
77
+ margin_percent: 边缘裁剪比例(去除可能的边框)
78
+
79
+ Returns:
80
+ 排序好的 (x, y, w, h) 列表
81
+ """
82
+ h_img, w_img = img.shape[:2]
83
+
84
+ # 去除边缘
85
+ margin_x = int(w_img * margin_percent)
86
+ margin_y = int(h_img * margin_percent)
87
+
88
+ effective_width = w_img - 2 * margin_x
89
+ effective_height = h_img - 2 * margin_y
90
+
91
+ # 计算每个单元格的尺寸
92
+ cell_width = effective_width // expected_cols
93
+ cell_height = effective_height // expected_rows
94
+
95
+ boxes = []
96
+ for row in range(expected_rows):
97
+ for col in range(expected_cols):
98
+ x = margin_x + col * cell_width
99
+ y = margin_y + row * cell_height
100
+ boxes.append((x, y, cell_width, cell_height))
101
+
102
+ return boxes
103
+
104
+ def detect_grid_cells_with_lines(img, expected_cols=6, expected_rows=4):
105
+ """
106
+ 通过形态学操作检测网格线,并提取每个格子的坐标
107
+ """
108
+ gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)
109
+
110
+ # 二值化 (反转:背景黑,内容/线白)
111
+ # 使用自适应阈值来应对光照或颜色不均
112
+ thresh = cv2.adaptiveThreshold(gray, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C,
113
+ cv2.THRESH_BINARY_INV, 11, 2)
114
+
115
+ # 定义结构元素 (Kernel) - 增大kernel以更好地检测断裂的线
116
+ h_img, w_img = img.shape[:2]
117
+ # 水平线 Kernel: 宽度长,高度为1
118
+ horizontal_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (w_img // 15, 1))
119
+ # 垂直线 Kernel: 宽度为1,高度长
120
+ vertical_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (1, h_img // 15))
121
+
122
+ # 1. 提取水平线
123
+ detect_horizontal = cv2.morphologyEx(thresh, cv2.MORPH_OPEN, horizontal_kernel, iterations=2)
124
+
125
+ # 2. 提取垂直线
126
+ detect_vertical = cv2.morphologyEx(thresh, cv2.MORPH_OPEN, vertical_kernel, iterations=2)
127
+
128
+ # 3. 合并网格线
129
+ grid_mask = cv2.addWeighted(detect_horizontal, 0.5, detect_vertical, 0.5, 0)
130
+ _, grid_mask = cv2.threshold(grid_mask, 0, 255, cv2.THRESH_BINARY)
131
+
132
+ # 更强的膨胀操作,连接断裂的网格线
133
+ kernel_dilate = cv2.getStructuringElement(cv2.MORPH_RECT, (5, 5))
134
+ grid_mask = cv2.dilate(grid_mask, kernel_dilate, iterations=3)
135
+
136
+ # 闭运算,进一步连接断裂
137
+ kernel_close = cv2.getStructuringElement(cv2.MORPH_RECT, (7, 7))
138
+ grid_mask = cv2.morphologyEx(grid_mask, cv2.MORPH_CLOSE, kernel_close, iterations=2)
139
+
140
+ # 4. 寻找所有的“洞”(即单元格)
141
+ # 我们通过查找 grid_mask 的轮廓,通常很难直接找到内部的矩形。
142
+ # 更好的方法是:找出网格��轮廓,画在全黑背景上,然后寻找连通组件,或者反转图片找白色方块。
143
+
144
+ # 这里我们采用“反转 mask”法:网格线是黑,格子是白
145
+ contours_mask = cv2.bitwise_not(grid_mask)
146
+ contours, _ = cv2.findContours(contours_mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)
147
+
148
+ # 计算期望的单元格面积
149
+ expected_cell_area = (w_img * h_img) / (expected_cols * expected_rows)
150
+
151
+ # 过滤微小的噪点轮廓,同时也过滤过大的轮廓
152
+ # 放宽过滤条件以捕获更多单元格
153
+ min_area = expected_cell_area / 20 # 单元格面积的 1/20 (之前是1/10)
154
+ max_area = expected_cell_area * 3 # 单元格面积的3倍 (之前是2倍)
155
+
156
+ # 调试信息
157
+ print(f" [调试] 图像尺寸: {w_img}x{h_img}, 检测到轮廓: {len(contours)}个")
158
+ print(f" [调试] 期望单元格面积: {expected_cell_area:.0f}, 过滤范围: {min_area:.0f}-{max_area:.0f}")
159
+
160
+ # 统计被过滤掉的轮廓
161
+ filtered_out = []
162
+ valid_contours = []
163
+ for c in contours:
164
+ area = cv2.contourArea(c)
165
+ if min_area < area < max_area:
166
+ valid_contours.append(c)
167
+ else:
168
+ x, y, w, h = cv2.boundingRect(c)
169
+ filtered_out.append((area, x, y, w, h))
170
+
171
+ if filtered_out:
172
+ print(f" [调试] 被过滤掉 {len(filtered_out)} 个轮廓:")
173
+ for area, x, y, w, h in sorted(filtered_out, key=lambda t: t[0], reverse=True)[:5]:
174
+ print(f" - 面积={area:.0f}, 位置=({x},{y}), 尺寸={w}x{h}")
175
+
176
+ # 排序:确保顺序是 左->右,上->下
177
+ sorted_boxes, row_counts = sort_contours(valid_contours)
178
+
179
+ return sorted_boxes, row_counts
180
+
181
+ def detect_grid_cells(img, expected_cols=6, expected_rows=4):
182
+ """
183
+ 鲁棒的网格检测方法:首先尝试检测网格线,如果失败则使用均匀分割
184
+ """
185
+ expected_count = expected_cols * expected_rows
186
+
187
+ # 方法1: 尝试检测网格线
188
+ sorted_boxes, row_counts = detect_grid_cells_with_lines(img, expected_cols, expected_rows)
189
+
190
+ # 严格检查:必须恰好检测到期望数量的单元格
191
+ if len(sorted_boxes) != expected_count:
192
+ print(f" 网格线检测不理想(检测到 {len(sorted_boxes)} 个单元格,期望 {expected_count} 个)")
193
+ if row_counts:
194
+ row_info = ", ".join([f"第{i+1}行: {count}个" for i, count in enumerate(row_counts)])
195
+ print(f" 检测到的行分布: {row_info}")
196
+ print(f" 切换到均匀分割模式...")
197
+ sorted_boxes = uniform_grid_split(img, expected_cols, expected_rows)
198
+ # 均匀分割时,打印每行的单元格数
199
+ print(f" 均匀分割结果:每行 {expected_cols} 个单元格,共 {expected_rows} 行")
200
+ else:
201
+ print(f" ✓ 成功检测到 {len(sorted_boxes)} 个网格单元格(符合预期)")
202
+ # 打印每行的单元格数量
203
+ if row_counts:
204
+ row_info = ", ".join([f"第{i+1}行: {count}个" for i, count in enumerate(row_counts)])
205
+ print(f" 行分布: {row_info}")
206
+
207
+ return sorted_boxes
208
+
209
+ def is_likely_text_region(img_region, thresh_region):
210
+ """
211
+ 判断一个区域是否可能是文字
212
+ 文字的特征:
213
+ 1. 主要是黑色或深色
214
+ 2. 高度较小
215
+ 3. 像素密度适中(不是纯色块)
216
+ """
217
+ if img_region.shape[0] == 0 or img_region.shape[1] == 0:
218
+ return False
219
+
220
+ # 转换为灰度(如果不是)
221
+ if len(img_region.shape) == 3:
222
+ gray_region = cv2.cvtColor(img_region, cv2.COLOR_BGR2GRAY)
223
+ else:
224
+ gray_region = img_region
225
+
226
+ # 检查1:高度不能太大(文字通常较矮)
227
+ height_ratio = img_region.shape[0] / img_region.shape[1] if img_region.shape[1] > 0 else 1
228
+ if height_ratio > 0.3: # 如果高度超过宽度的30%,可能不是单行文字
229
+ return False
230
+
231
+ # 检查2:颜色是否偏暗(文字通常是黑色或深色)
232
+ mean_brightness = np.mean(gray_region)
233
+ if mean_brightness > 200: # 太亮,不像文字
234
+ return False
235
+
236
+ # 检查3:内容像素占比(文字不会太密集也不会太稀疏)
237
+ content_pixels = np.sum(thresh_region > 0)
238
+ total_pixels = thresh_region.shape[0] * thresh_region.shape[1]
239
+ density = content_pixels / total_pixels if total_pixels > 0 else 0
240
+
241
+ if density < 0.05 or density > 0.5: # 密度不在合理范围
242
+ return False
243
+
244
+ return True
245
+
246
+ def detect_and_remove_text(img, thresh, row_sums):
247
+ """
248
+ 检测并移除图标上方或下方的文字标题
249
+
250
+ 返回: (top_crop, bottom_crop) - 需要裁剪的上下边界
251
+ """
252
+ h = len(row_sums)
253
+
254
+ # 定义"空白行"的阈值(行和很小)
255
+ empty_threshold = max(5, img.shape[1] * 0.01) # 至少5,或宽度的1%
256
+ # 定义"间隙"的最小行数
257
+ min_gap_rows = max(2, int(h * 0.02)) # 至少2行,或高度的2%
258
+
259
+ # 找到所有内容行(非空白行)
260
+ content_rows = [i for i, val in enumerate(row_sums) if val > empty_threshold]
261
+
262
+ if len(content_rows) == 0:
263
+ return 0, h
264
+
265
+ # 找到主要内容区域(最大的连续内容块)
266
+ # 先找出所有的间隙
267
+ gaps = []
268
+ if len(content_rows) > 1:
269
+ for i in range(len(content_rows) - 1):
270
+ gap_size = content_rows[i + 1] - content_rows[i] - 1
271
+ if gap_size >= min_gap_rows:
272
+ gap_start = content_rows[i]
273
+ gap_end = content_rows[i + 1]
274
+ gaps.append((gap_start, gap_end, gap_size))
275
+
276
+ top_crop = 0
277
+ bottom_crop = h
278
+
279
+ # 如果存在明显的间隙,说明可能有分离的文字
280
+ if gaps:
281
+ # 找到最大的间隙
282
+ largest_gap = max(gaps, key=lambda x: x[2])
283
+ gap_start, gap_end, gap_size = largest_gap
284
+
285
+ # 计算间隙上方和下方的内容量和行数
286
+ top_rows = gap_start
287
+ bottom_rows = h - gap_end
288
+ top_content = sum(row_sums[:gap_start])
289
+ bottom_content = sum(row_sums[gap_end:])
290
+
291
+ # 判断哪一部分是主要图标,哪一部分是文字
292
+ # 文字的特征:1) 内容较少 2) 行数较少 3) 符合文字特征
293
+
294
+ # 检查上方区域
295
+ if top_rows > 0 and top_rows < h * 0.3: # 上方行数不超过30%
296
+ if top_content < bottom_content * 0.4: # 上方内容明显少于下方
297
+ # 进一步检查是否像文字
298
+ top_region = img[:gap_start, :]
299
+ top_thresh = thresh[:gap_start, :]
300
+ if is_likely_text_region(top_region, top_thresh):
301
+ top_crop = gap_end
302
+
303
+ # 检查下方区域
304
+ if bottom_rows > 0 and bottom_rows < h * 0.3: # 下方行数不超过30%
305
+ if bottom_content < top_content * 0.4: # 下方内容明显少于上方
306
+ # 进一步检查是否像文字
307
+ bottom_region = img[gap_end:, :]
308
+ bottom_thresh = thresh[gap_end:, :]
309
+ if is_likely_text_region(bottom_region, bottom_thresh):
310
+ bottom_crop = gap_start
311
+
312
+ return top_crop, bottom_crop
313
+
314
+ def smart_crop_icon(img, padding=10):
315
+ """
316
+ 单个 Icon 处理:去字、去空、加 Padding
317
+ 增强版:可以检测并删除上方或下方的文字标题
318
+ """
319
+ h, w = img.shape[:2]
320
+
321
+ # 1. 裁剪掉可能残留的网格边缘 (比如四周切掉 3px)
322
+ margin = 3
323
+ if h > 2*margin and w > 2*margin:
324
+ img = img[margin:-margin, margin:-margin]
325
+ h, w = img.shape[:2]
326
+
327
+ gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)
328
+ _, thresh = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)
329
+
330
+ # 2. 使用改进的方法检测并移除文字
331
+ row_sums = np.sum(thresh, axis=1)
332
+ top_crop, bottom_crop = detect_and_remove_text(img, thresh, row_sums)
333
+
334
+ # 应用裁剪
335
+ if top_crop > 0 or bottom_crop < h:
336
+ img = img[top_crop:bottom_crop, :]
337
+ thresh = thresh[top_crop:bottom_crop, :]
338
+
339
+ # 3. 寻找 Icon 的精确边界
340
+ coords = cv2.findNonZero(thresh)
341
+ if coords is not None:
342
+ x, y, w_box, h_box = cv2.boundingRect(coords)
343
+
344
+ # 裁剪并添加 Padding
345
+ # 创建一个新的白色画布
346
+ final_h = h_box + 2 * padding
347
+ final_w = w_box + 2 * padding
348
+ canvas = np.ones((final_h, final_w, 3), dtype=np.uint8) * 255
349
+
350
+ # 提取 icon 内容
351
+ icon_content = img[y:y+h_box, x:x+w_box]
352
+
353
+ # 将 icon 贴到画布中心
354
+ canvas[padding:padding+h_box, padding:padding+w_box] = icon_content
355
+ return canvas
356
+
357
+ return img
358
+
359
+ def process_image_robust(image_path, labels_data):
360
+ if not os.path.exists(image_path):
361
+ print(f"Error: {image_path} not found.")
362
+ return
363
+
364
+ print(f"Processing: {image_path} ...")
365
+ img = cv2.imread(image_path)
366
+
367
+ # 1. 检测网格
368
+ # 返回的是排序好的 (x, y, w, h) 列表
369
+ grid_boxes = detect_grid_cells(img, expected_cols=6, expected_rows=4)
370
+
371
+ # 2. 准备文本数据
372
+ lines = [l.strip() for l in labels_data.strip().split('\n') if l.strip()]
373
+ style = "flat"
374
+ start_idx = 0
375
+ if lines[0].lower().startswith("style:"):
376
+ style = clean_filename(lines[0].split(':')[1])
377
+ start_idx = 1
378
+
379
+ output_dir = "extracted_icons"
380
+ if not os.path.exists(output_dir):
381
+ os.makedirs(output_dir)
382
+
383
+ # 3. 遍历并保存
384
+ for i, box in enumerate(grid_boxes):
385
+ text_idx = start_idx + i
386
+ if text_idx >= len(lines):
387
+ break
388
+
389
+ # 解析文本
390
+ line_text = lines[text_idx]
391
+ parts = line_text.split(',', 1)
392
+ if len(parts) == 2:
393
+ category = clean_filename(parts[0])
394
+ name = clean_filename(parts[1])
395
+ else:
396
+ category = "icon"
397
+ name = clean_filename(parts[0])
398
+
399
+ filename = f"{category}-{name}-{style}.png"
400
+ save_path = os.path.join(output_dir, filename)
401
+
402
+ # 提取单元格
403
+ x, y, w, h = box
404
+ cell_img = img[y:y+h, x:x+w]
405
+
406
+ # 智能裁切
407
+ final_img = smart_crop_icon(cell_img, padding=10)
408
+
409
+ cv2.imwrite(save_path, final_img)
410
+ # print(f"Saved: {filename}") # 减少刷屏
411
+
412
+ print(f"Done. Extracted {len(grid_boxes)} icons to '{output_dir}/'.\n")
413
+
414
+ def process_batch_range(start_batch, end_batch, base_dir="generated_icons"):
415
+ """
416
+ 批量处理指定范围内的batch文件夹下的所有png文件
417
+
418
+ Args:
419
+ start_batch: 起始batch编号 (例如: 1)
420
+ end_batch: 结束batch编号 (例如: 10)
421
+ base_dir: batch文件夹所在的基础目录
422
+ """
423
+ print(f"开始批量处理 batch_{start_batch:04d} 到 batch_{end_batch:04d} ...\n")
424
+
425
+ total_processed = 0
426
+ failed_files = []
427
+
428
+ for batch_num in range(start_batch, end_batch + 1):
429
+ batch_dir = os.path.join(base_dir, f"batch_{batch_num:04d}")
430
+
431
+ # 检查batch文件夹是否存在
432
+ if not os.path.exists(batch_dir):
433
+ print(f"Warning: {batch_dir} 不存在,跳过...")
434
+ continue
435
+
436
+ print(f"处理 {batch_dir} ...")
437
+
438
+ # 查找该batch下的所有png文件
439
+ png_files = glob.glob(os.path.join(batch_dir, "*.png"))
440
+
441
+ if not png_files:
442
+ print(f" 未找到png文件,跳过...")
443
+ continue
444
+
445
+ # 处理每个png文件
446
+ for png_path in sorted(png_files):
447
+ # 构造对应的txt文件路径
448
+ txt_path = png_path.rsplit('.', 1)[0] + '.txt'
449
+
450
+ # 检查txt文件是否存在
451
+ if not os.path.exists(txt_path):
452
+ print(f" Warning: {txt_path} 不存在,跳过 {os.path.basename(png_path)}")
453
+ failed_files.append(png_path)
454
+ continue
455
+
456
+ # 读取txt文件内容
457
+ try:
458
+ with open(txt_path, 'r', encoding='utf-8') as f:
459
+ labels_data = f.read()
460
+
461
+ # 处理图像
462
+ process_image_robust(png_path, labels_data)
463
+ total_processed += 1
464
+
465
+ except Exception as e:
466
+ print(f" Error processing {os.path.basename(png_path)}: {str(e)}")
467
+ failed_files.append(png_path)
468
+
469
+ # 输出总结
470
+ print(f"\n{'='*60}")
471
+ print(f"批量处理完成!")
472
+ print(f"成功处理: {total_processed} 个文件")
473
+
474
+ if failed_files:
475
+ print(f"失败/跳过: {len(failed_files)} 个文件")
476
+ print("失败文件列表:")
477
+ for f in failed_files:
478
+ print(f" - {f}")
479
+ print(f"{'='*60}")
480
+
481
+ if __name__ == "__main__":
482
+ # 设置要处理的batch范围
483
+ START_BATCH = 1 # 起始batch编号
484
+ END_BATCH = 200 # 结束batch编号
485
+
486
+ # 执行批量处理
487
+ process_batch_range(START_BATCH, END_BATCH)
icon_generation/template.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "minimal_flat": "Create a minimal flat icon representing {DOMAIN}: {SPECIFIC ATTRIBUTE}. Use simple flat shapes with solid colors, no outlines, no gradients, no shadows, no depth, no texture. Maintain a clean, modern look with clear geometry and a white background.",
3
+ "outline_line": "Create an outline icon representing {DOMAIN}: {SPECIFIC ATTRIBUTE}. Use clean, uniform stroke lines only, no fills, no shading, no texture, no gradients. Keep the design light, airy, and minimal, with a white background.",
4
+ "solid_filled": "Create a solid filled icon representing {DOMAIN}: {SPECIFIC ATTRIBUTE}. Use bold, fully filled shapes with strong silhouettes, no outlines, no gradients, no texture, no shading. High contrast, simple form, white background.",
5
+ "simplified_cartoon": "Create a simplified cartoon-style icon representing {DOMAIN}: {SPECIFIC ATTRIBUTE}. Use rounded shapes, playful proportions, and simplified details. Avoid realism, texture, or shading. Keep the icon friendly, colorful, and clean on a white background.",
6
+ "hand_drawn_sketch": "Create a hand-drawn sketch-style icon representing {DOMAIN}: {SPECIFIC ATTRIBUTE}. Use rough, imperfect strokes with visible hand-drawn variation. No straight mechanical lines, no fills, minimal shading, organic and expressive lines on a white background.",
7
+ "doodle": "Create a doodle-style icon representing {DOMAIN}: {SPECIFIC ATTRIBUTE}. Use loose, casual, spontaneous linework with playful and whimsical energy. Keep it simple and informal, no precise geometry, no fills, no shading, white background.",
8
+ "isometric": "Create an isometric icon representing {DOMAIN}: {SPECIFIC ATTRIBUTE}. Use a fixed isometric perspective with simple 3D-like forms. Apply flat colors with subtle separation between surfaces, no realistic lighting, no heavy shadows, clean and structured on a white background.",
9
+ "pictogram": "Create a pictogram icon representing {DOMAIN}: {SPECIFIC ATTRIBUTE}. Use highly simplified, universally recognizable shapes with clear symbolism. Avoid decorative details, textures, or perspective. Flat, functional design on a white background.",
10
+ "glyph": "Create a glyph icon representing {DOMAIN}: {SPECIFIC ATTRIBUTE}. Use extremely minimal, symbol-like forms designed for clarity at small sizes. Solid shapes only, no details, no texture, no shading, monochrome on a white background.",
11
+ "pixel": "Create a pixel-style icon representing {DOMAIN}: {SPECIFIC ATTRIBUTE}. Use a visible pixel grid, low resolution, and blocky shapes. Avoid smooth curves, gradients, or anti-aliasing. Retro digital style on a plain background.",
12
+ "chalkboard": "Create a chalkboard-style icon representing {DOMAIN}: {SPECIFIC ATTRIBUTE}. Use chalk-like textured strokes with slightly uneven edges. White or light chalk lines on a dark chalkboard background. No smooth vector lines, educational and hand-drawn feel.",
13
+ "neon_glow": "Create a neon glow icon representing {DOMAIN}: {SPECIFIC ATTRIBUTE}. Use bright glowing outlines with a soft neon light effect. Dark background, high contrast, futuristic or nightlife aesthetic. Avoid flat colors or solid fills."
14
+ }
icon_generation/template_batch.json ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "minimal_flat": "Create a grid of minimal flat icons. Each icon represents a specific domain-attribute pair from the following list: {DOMAIN_ATTRIBUTE_PAIRS}. Use a clear and strict grid layout with thin black divider lines separating each icon. Each icon should have equal size and consistent alignment. Use simple flat shapes with solid colors, no outlines, no gradients, no shadows, no depth, no texture. Clean, modern geometry.",
3
+
4
+ "outline_line": "Create a grid of colored outline icons. Each icon represents a specific domain-attribute pair from the following list: {DOMAIN_ATTRIBUTE_PAIRS}. Use a strict grid layout with thin black divider lines between icons. All icons must be uniform in size and spacing. Use clean, consistent colored stroke lines only. No fills, no shading, no texture, no gradients. Icons should feature simple outlines rendered in color, not black.",
5
+
6
+ "solid_filled": "Create a grid of solid filled icons. Each icon represents a specific domain-attribute pair from the following list: {DOMAIN_ATTRIBUTE_PAIRS}. Arrange icons in a uniform grid separated by thin black lines. Each icon should have a strong silhouette, fully filled shapes, no outlines, no gradients, no texture, no shading. High contrast, simple forms.",
7
+
8
+ "simplified_cartoon": "Create a grid of simplified cartoon-style icons. Each icon represents a specific domain-attribute pair from the following list: {DOMAIN_ATTRIBUTE_PAIRS}. Use a consistent grid layout with thin black divider lines. Icons must be equal in size and evenly spaced. Use rounded shapes, playful proportions, simplified details. No realism, no texture, no shading.",
9
+
10
+ "hand_drawn_sketch": "Create a grid of hand-drawn sketch-style icons. Each icon represents a specific domain-attribute pair from the following list: {DOMAIN_ATTRIBUTE_PAIRS}. Display icons in a structured grid separated by thin black lines. Maintain consistent icon size despite sketch variation. Use rough, imperfect strokes with visible hand-drawn character. Apply appropriate colorful fills to each icon, while keeping minimal shading and organic lines.",
11
+
12
+ "doodle": "Create a grid of doodle-style icons. Each icon represents a specific domain-attribute pair from the following list: {DOMAIN_ATTRIBUTE_PAIRS}. Arrange icons in a clear grid with thin black divider lines. Keep all icons evenly sized and aligned. Use loose, playful, spontaneous linework with a casual feel, and add colorful fills to each icon. No shading, no precise geometry.",
13
+
14
+ "isometric": "Create a grid of isometric icons. Each icon represents a specific domain-attribute pair from the following list: {DOMAIN_ATTRIBUTE_PAIRS}. Use a strict grid layout with thin black divider lines between icons. Maintain consistent scale and isometric angle across all icons. Use flat colors with simple surface separation, no realistic lighting, no heavy shadows.",
15
+
16
+ "glyph": "Create a grid of glyph icons. Each icon represents a specific domain-attribute pair from the following list: {DOMAIN_ATTRIBUTE_PAIRS}. Use a precise grid layout with thin black divider lines. All glyphs must be equal in size and visually balanced. Use extremely minimal, symbol-like solid forms with minimal coloring—use just a few appropriate colors sparingly for clarity or emphasis. No details, no texture, no shading. Avoid monochrome.",
17
+
18
+ "chalkboard": "Create a grid of colorful chalk-style icons. Each icon represents a specific domain-attribute pair from the following list: {DOMAIN_ATTRIBUTE_PAIRS}. Use a fixed grid layout separated by thin black divider lines. All icons should be consistently sized. Use multicolored chalk-like textured strokes with slightly uneven edges, vibrant pastel colors, and a visible hand-drawn look. Adapt for high visibility and contrast on a white background.",
19
+
20
+ "neon_glow": "Create a grid of neon-style glow icons. Each icon represents a specific domain-attribute pair from the following list: {DOMAIN_ATTRIBUTE_PAIRS}. Arrange icons in a strict grid with thin black divider lines. Icons should be uniform in size and alignment. Use bright glowing outlines and soft neon effects adapted for high contrast on a white background."
21
+ }
icon_generation/test_split.py ADDED
@@ -0,0 +1,56 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """
3
+ 测试split_icon.py的网格检测功能
4
+ """
5
+ import cv2
6
+ from split_icon import detect_grid_cells, uniform_grid_split
7
+ import os
8
+
9
+ def test_image(image_path):
10
+ """测试单个图像的网格检测"""
11
+ if not os.path.exists(image_path):
12
+ print(f"错误:文件不存在 {image_path}")
13
+ return
14
+
15
+ print(f"\n{'='*60}")
16
+ print(f"测试图像: {os.path.basename(image_path)}")
17
+ print(f"{'='*60}")
18
+
19
+ # 读取图像
20
+ img = cv2.imread(image_path)
21
+ if img is None:
22
+ print("错误:无法读取图像")
23
+ return
24
+
25
+ print(f"图像尺寸: {img.shape[1]}x{img.shape[0]}")
26
+
27
+ # 测试网格检测
28
+ boxes = detect_grid_cells(img, expected_cols=6, expected_rows=4)
29
+ print(f"最终检测到的单元格数量: {len(boxes)}")
30
+
31
+ # 可视化结果
32
+ result_img = img.copy()
33
+ for i, box in enumerate(boxes):
34
+ x, y, w, h = box
35
+ cv2.rectangle(result_img, (x, y), (x+w, y+h), (0, 255, 0), 2)
36
+ # 在左上角添加序号
37
+ cv2.putText(result_img, str(i+1), (x+5, y+20),
38
+ cv2.FONT_HERSHEY_SIMPLEX, 0.5, (255, 0, 0), 2)
39
+
40
+ # 保存结果
41
+ output_path = image_path.rsplit('.', 1)[0] + '_detected.png'
42
+ cv2.imwrite(output_path, result_img)
43
+ print(f"检测结果已保存到: {output_path}")
44
+
45
+ if __name__ == "__main__":
46
+ # 测试batch_0001中的一张图像
47
+ test_images = [
48
+ "generated_icons/batch_0001/batch_0001_doodle.png",
49
+ "generated_icons/batch_0001/batch_0001_hand_drawn_sketch.png",
50
+ "generated_icons/batch_0001/batch_0001_isometric.png",
51
+ ]
52
+
53
+ for img_path in test_images:
54
+ if os.path.exists(img_path):
55
+ test_image(img_path)
56
+ break