Spaces:
Running on CPU Upgrade
Running on CPU Upgrade
| import json | |
| import os | |
| import re | |
| import pandas as pd | |
| from datetime import datetime | |
| from utils import create_hyperlinked_names, process_model_size, MODEL_SIZE_COL_NAME | |
| from datasets import * | |
| BASE_COLS = ['Rank', 'Models', MODEL_SIZE_COL_NAME, 'Date'] | |
| BASE_DATA_TITLE_TYPE = ['str', 'markdown', 'str', 'str'] | |
| OVERALL_COLS_V2 = ["Overall-V2", 'Image-Overall(V2)', 'Video-Overall', 'Visdoc-Overall'] | |
| COLUMN_NAMES_V2 = BASE_COLS + OVERALL_COLS_V2 | |
| DATA_TITLE_TYPE_V2 = BASE_DATA_TITLE_TYPE + \ | |
| ['number'] * len(OVERALL_COLS_V2) | |
| OVERALL_COLS_V3 = ["Overall", "Overall-V3π", "Text-Overall", "Audio-Overall", "Agent-Overall", 'Image-Overall', 'Video-Overall', 'Visdoc-Overall'] | |
| COLUMN_NAMES_V3 = BASE_COLS + OVERALL_COLS_V3 | |
| DATA_TITLE_TYPE_V3 = BASE_DATA_TITLE_TYPE + \ | |
| ['number'] * len(OVERALL_COLS_V3) | |
| SUB_TASKS_T = ["FollowIR", "R2MED", "InfoSearch", "BRIGHT", "LongEmbed", "MultiConIR", "NanoBEIR"] | |
| TASKS_T = ['Text-Overall'] + SUB_TASKS_T + ALL_DATASETS_SPLITS['text'] | |
| COLUMN_NAMES_T = BASE_COLS + TASKS_T | |
| DATA_TITLE_TYPE_T = BASE_DATA_TITLE_TYPE + \ | |
| ['number'] * len(TASKS_T) | |
| SUB_TASKS_I = ["I-CLS", "I-QA", "I-RET", "I-VG"] | |
| TASKS_I = ['Image-Overall', 'Image-Overall(V2)'] + SUB_TASKS_I + ALL_DATASETS_SPLITS['image'] | |
| COLUMN_NAMES_I = BASE_COLS + TASKS_I | |
| DATA_TITLE_TYPE_I = BASE_DATA_TITLE_TYPE + \ | |
| ['number'] * len(TASKS_I) | |
| SUB_TASKS_V = ["V-CLS", "V-QA", "V-RET", "V-MRET"] | |
| TASKS_V = ['Video-Overall'] + SUB_TASKS_V + ALL_DATASETS_SPLITS['video'] | |
| COLUMN_NAMES_V = BASE_COLS + TASKS_V | |
| DATA_TITLE_TYPE_V = BASE_DATA_TITLE_TYPE + \ | |
| ['number'] * len(TASKS_V) | |
| SUB_TASKS_A = ["A-CLS", "A-RET"] | |
| TASKS_A = ['Audio-Overall'] + SUB_TASKS_A + ALL_DATASETS_SPLITS['audio'] | |
| COLUMN_NAMES_A = BASE_COLS + TASKS_A | |
| DATA_TITLE_TYPE_A = BASE_DATA_TITLE_TYPE + \ | |
| ['number'] * len(TASKS_A) | |
| SUB_TASKS_D = ['ViDoRe-V1', 'ViDoRe-V2', 'VisRAG', 'VisDoc-OOD'] | |
| TASKS_D = ['Visdoc-Overall'] + SUB_TASKS_D + ALL_DATASETS_SPLITS['visdoc'] | |
| COLUMN_NAMES_D = BASE_COLS + TASKS_D | |
| DATA_TITLE_TYPE_D = BASE_DATA_TITLE_TYPE + \ | |
| ['number'] * len(TASKS_D) | |
| SUB_TASKS_AG = ['Tool', 'GUI', 'Memory'] | |
| TASKS_AG = ['Agent-Overall'] + SUB_TASKS_AG + ALL_DATASETS_SPLITS['agent'] | |
| COLUMN_NAMES_AG = BASE_COLS + TASKS_AG | |
| DATA_TITLE_TYPE_AG = BASE_DATA_TITLE_TYPE + \ | |
| ['number'] * len(TASKS_AG) | |
| TABLE_INTRODUCTION = """**MMEB**: Massive MultiModal Embedding Benchmark \n | |
| Models are ranked based on **Overall**(V3-ALL). **Overall-V3π**: Newly added datasets in V3 (e.g., agent, audio, text and MCMR)""" | |
| TABLE_INTRODUCTION_I = """**I-CLS**: Image Classification, **I-QA**: (Image) Visual Question Answering, **I-RET**: Image Retrieval, **I-VG**: (Image) Visual Grounding \n | |
| Models are ranked based on **Image-Overall**, the V3 version which includes the newly added I-RET dataset MCMR.\n | |
| **Image-Overall(V2)** is the original V2 version that excludes MCMR.\n""" | |
| TABLE_INTRODUCTION_V = """**V-CLS**: Video Classification, **V-QA**: (Video) Visual Question Answering, **V-RET**: Video Retrieval, **V-MRET**: Video Moment Retrieval \n | |
| Models are ranked based on **Video-Overall**""" | |
| TABLE_INTRODUCTION_A = """**A-CLS**: Audio Classification, **A-RET**: Audio Retrieval \n | |
| Models are ranked based on **Audio-Overall**""" | |
| TABLE_INTRODUCTION_D = """β οΈ Please re-evaluate your models if you see a 0 on ViDoSeek-page-fixed or MMLongBench-page-fixed datasets. \n | |
| **VisDoc**: Visual Document Understanding \n | |
| Models are ranked based on **Visdoc-Overall**""" | |
| TABLE_INTRODUCTION_AG = """**Tool**: Tool Retrieval, **GUI**: GUI Control, **Memory**: Agent Memory Retrieval \n | |
| Models are ranked based on **Agent-Overall**""" | |
| LEADERBOARD_INFO = """ | |
| ## Dataset Summary | |
| """ | |
| CITATION_BUTTON_TEXT_V2 = r"""@misc{meng2025vlm2vecv2advancingmultimodalembedding, | |
| title={VLM2Vec-V2: Advancing Multimodal Embedding for Videos, Images, and Visual Documents}, | |
| author={Rui Meng and Ziyan Jiang and Ye Liu and Mingyi Su and Xinyi Yang and Yuepeng Fu and Can Qin and Zeyuan Chen and Ran Xu and Caiming Xiong and Yingbo Zhou and Wenhu Chen and Semih Yavuz}, | |
| year={2025}, | |
| eprint={2507.04590}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.CV}, | |
| url={https://arxiv.org/abs/2507.04590}, | |
| }""" | |
| CITATION_BUTTON_TEXT_V3 = r"""@misc{huang2026mmebv3measuringperformancegaps, | |
| title={MMEB-V3: Measuring the Performance Gaps of Omni-Modality Embedding Models}, | |
| author={Haohang Huang and Xuan Lu and Mingyi Su and Xuan Zhang and Ziyan Jiang and Ping Nie and Kai Zou and Tomas Pfister and Wenhu Chen and Wei Zhang and Xiaoyu Shen and Rui Meng}, | |
| year={2026}, | |
| eprint={2604.23321}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.IR}, | |
| url={https://arxiv.org/abs/2604.23321}, | |
| }""" | |
| def load_single_json(file_path): | |
| with open(file_path, 'r') as file: | |
| data = json.load(file) | |
| return data | |
| def load_data(base_dir=SCORE_BASE_DIR): | |
| all_data = [] | |
| for file_name in os.listdir(base_dir): | |
| if file_name.endswith('.json'): | |
| file_path = os.path.join(base_dir, file_name) | |
| data = load_single_json(file_path) | |
| all_data.append(data) | |
| return all_data | |
| def load_scores(raw_scores={}): | |
| """This function loads the raw scores from the user provided scores summary and flattens them into a single dictionary.""" | |
| # merge the three agent tasks into one | |
| if any(_ in raw_scores for _ in ['tool', 'gui', 'memory']): | |
| raw_scores['agent'] = raw_scores.pop('tool', {}) | raw_scores.pop('gui', {}) | raw_scores.pop('memory', {}) | |
| all_scores = {} | |
| for modality, datasets_list in DATASETS.items(): # Ex.: ('image', {'I-CLS': [...], 'I-QA': [...]}) | |
| for sub_task, datasets in datasets_list.items(): # Ex.: ('I-CLS', ['VOC2007', 'N24News', ...]) | |
| for dataset in datasets: # Ex.: 'VOC2007' | |
| score = raw_scores.get(modality, {}).get(dataset, 0.0) | |
| score = 0.0 if isinstance(score, str) and "N/A" in score else score | |
| metric = SPECIAL_METRICS.get(dataset, 'hit@1') | |
| if isinstance(score, dict): | |
| if 'visdoc' in modality: | |
| metric = "ndcg_linear@5" if "ndcg_linear@5" in score else "ndcg@5" | |
| score = score.get(metric, 0.0) | |
| all_scores[dataset] = round(score * 100.0, 2) | |
| return all_scores | |
| def get_avg(sum_score, leng): | |
| avg = sum_score / leng if leng > 0 else 0.0 | |
| avg = round(avg, 2) # Round to 2 decimal places | |
| return avg | |
| def calculate_score(raw_scores=None): | |
| """This function calculates the overall average scores for all datasets as well as avg scores for each modality and sub-task based on the raw scores. | |
| """ | |
| all_scores = load_scores(raw_scores) | |
| avg_scores = {} | |
| # Calculate overall score for all datasets | |
| avg_scores['Overall'] = get_avg(sum(all_scores.values()), len(ALL_DATASETS)) | |
| v2_scores = {k:v for k,v in all_scores.items() if k in ALL_DATASETS_SPLITS['image'] or k in ALL_DATASETS_SPLITS['video'] or k in ALL_DATASETS_SPLITS['visdoc']} | |
| avg_scores['Overall-V2'] = get_avg(sum(v2_scores.values()), len(v2_scores)) | |
| v3_newonly_scores = {k:v for k,v in all_scores.items() if k in ALL_DATASETS_SPLITS['text'] or k in ALL_DATASETS_SPLITS['audio'] or k in ALL_DATASETS_SPLITS['agent'] or k == "MCMR"} | |
| avg_scores['Overall-V3π'] = get_avg(sum(v3_newonly_scores.values()), len(v3_newonly_scores)) | |
| # Calculate scores for each modality | |
| for modality in MODALITIES: | |
| datasets_for_each_modality = ALL_DATASETS_SPLITS[modality] | |
| avg_scores[f"{modality.capitalize()}-Overall"] = get_avg( | |
| sum(all_scores.get(dataset, 0.0) for dataset in datasets_for_each_modality), | |
| len(datasets_for_each_modality) | |
| ) | |
| # special process for image overall v2, excluding MCMR | |
| img_datasets_without_mcmr = [_ for _ in ALL_DATASETS_SPLITS["image"] if _ != "MCMR"] # exclude MCMR | |
| assert len(img_datasets_without_mcmr) == len(ALL_DATASETS_SPLITS["image"])-1 and "MCMR" not in img_datasets_without_mcmr, f"MCMR not removed properly" | |
| avg_scores["Image-Overall(V2)"] = get_avg(sum(all_scores.get(dataset, 0.0) for dataset in img_datasets_without_mcmr), len(img_datasets_without_mcmr)) | |
| # Calculate scores for each sub-task | |
| for modality, datasets_list in DATASETS.items(): | |
| for sub_task, datasets in datasets_list.items(): | |
| sub_task_score = sum(all_scores.get(dataset, 0.0) for dataset in datasets) | |
| avg_scores[sub_task] = get_avg(sub_task_score, len(datasets)) | |
| all_scores.update(avg_scores) | |
| return all_scores | |
| def generate_model_row(data): | |
| metadata = data['metadata'] | |
| row = { | |
| 'Models': metadata.get('model_name', None), | |
| MODEL_SIZE_COL_NAME: metadata.get('model_size', None), | |
| 'URL': metadata.get('url', None), | |
| 'Submitted by': metadata.get('data_source', 'Self-Reported'), | |
| 'Date': metadata.get('report_generated_date', None) | |
| } | |
| scores = calculate_score(data['metrics']) | |
| row.update(scores) | |
| return row | |
| def print_time(time: str|None): | |
| try: | |
| dt = datetime.strptime(time, "%Y-%m-%dT%H:%M:%S.%f") | |
| return dt.strftime("%y-%m-%d") | |
| except (ValueError, TypeError): | |
| return 'unknown' | |
| medal_map = { | |
| "1": "π", | |
| "2": "π₯", | |
| "3": "π₯" | |
| } | |
| def rank_models(df, column='Overall', rank_name='Rank'): | |
| """Ranks the models based on the specific score.""" | |
| df = df.sort_values(by=column, ascending=False).reset_index(drop=True) | |
| df[rank_name] = df[column].rank(method='min', ascending=False).astype(int).astype(str).map(lambda x: medal_map.get(x, x)) | |
| return df | |
| def get_df(rank_column='Overall'): | |
| """Generates a DataFrame from the loaded data.""" | |
| all_data = load_data() | |
| rows = [generate_model_row(data) for data in all_data] | |
| df = pd.DataFrame(rows) | |
| df[MODEL_SIZE_COL_NAME] = df[MODEL_SIZE_COL_NAME].apply(process_model_size) | |
| df['Date'] = df['Date'].apply(print_time) | |
| df = create_hyperlinked_names(df) | |
| df = rank_models(df, column=rank_column) | |
| return df | |
| def refresh_data(columns = COLUMN_NAMES_V3): | |
| df = get_df() | |
| return df[columns] | |
| def search_and_filter_models(df, query, min_size, max_size, columns = COLUMN_NAMES_V3): | |
| filtered_df = df.copy() | |
| if query: | |
| filtered_df = filtered_df[filtered_df['Models'].str.contains(query, case=False, na=False)] | |
| size_mask = filtered_df[MODEL_SIZE_COL_NAME].apply(lambda x: | |
| (min_size <= 1000.0 <= max_size) if x == 'unknown' | |
| else (min_size <= x <= max_size)) | |
| filtered_df = filtered_df[size_mask] | |
| return filtered_df[columns] | |
| def extract_link_data(text): | |
| # Regex to capture content inside href="..." and inside <a>...</a> | |
| pattern = r'href="([^"]+)".*?>(.*?)</a>' | |
| match = re.search(pattern, text) | |
| if match: | |
| # Found HTML: return (URL, Name) | |
| return match.group(1), match.group(2) | |
| # No HTML found: return (None, Original String) | |
| return None, text | |
| def save_ranking_summary(df, name, save_now=True, dir='rankings'): | |
| # df[['url', 'name']] = df['Models'].apply(lambda x: pd.Series(extract_link_data(x))) | |
| csv_path, json_path = os.path.join(dir, f'{name}.csv'), os.path.join(dir, f'{name}.jsonl') | |
| if save_now: | |
| df.to_csv(csv_path, index=False) | |
| df.to_json(json_path, orient='records', lines=True) | |
| return csv_path, json_path | |
| def download_ranking(df, name, format='csv', dir='rankings'): | |
| csv_path, json_path = save_ranking_summary(df, name, save_now=False, dir=dir) | |
| return csv_path if format == 'csv' else json_path | |