import os import re import sys import json import time import random import string import argparse import pandas as pd import copy import logging from tqdm import tqdm import PIL from PIL import ImageFile PIL.ImageFile.LOAD_TRUNCATED_IMAGES = True BASE_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) sys.path.append(BASE_DIR) from approach.pipeline_utils import ( enrich_ape_results, output_path_for_selection, select_jsonl_lines, write_json_atomic, ) os.environ['TORCH_HOME'] = './.cache' os.environ['HF_HOME'] = './.cache' vlm = 'llava7b' llm = 'gpt_3.5_turbo' ovod = 'grounding_dino' vlm_prompt = '' ### Configurable host_device = os.getenv('ORIENTER_LEGACY_HOST', 'local') perspective = 'all_perspective' # perspective = 'direct_back' # perspective = 'direct_front' # perspective = 'direct_side' # perspective = 'direct_top' # perspective = 'eyelevel' # perspective = 'overlook' ### Configurable llava_path = os.path.join(BASE_DIR, 'approach/vlm/LLaVA/llava/eval') grounding_dino_path = os.path.join(BASE_DIR, 'approach/ovod/GroundingDINO') ape_path = os.path.join(BASE_DIR, 'approach/ovod/APE') # TODO: data on the CUHK server # images_dir = os.path.join(BASE_DIR, 'dataset/data/coco_merged/images/interactable') # icse # images_dir = os.path.join(BASE_DIR, 'dataset/data/coco_merged/images/test_set_merged') # icse_rebuttal # images_dir = os.path.join(BASE_DIR, 'dataset/data/coco_angleview/images/all_perspective') # fse images_dir = os.path.join(BASE_DIR, 'dataset/data/coco_det/images/union3') # def generate_grounding_dino_command(ovod_image_path, ovod_output_dir, ovod_candidates): # gdino_object_str = ' . '.join(ovod_candidates) # gdino_command = f''' # python demo/inference_on_a_image.py \ # -c groundingdino/config/GroundingDINO_SwinT_OGC.py \ # -p weights/groundingdino_swint_ogc.pth \ # -i {ovod_image_path} \ # -o "{ovod_output_dir}" \ # -t "{gdino_object_str}" # ''' # return gdino_command def generate_question_file(img_folder, dst_path): # Ensure the directory exists if not os.path.exists(img_folder): print(f"Error: Directory {img_folder} does not exist.") return # List all files in the directory all_files = os.listdir(img_folder) # Filter out files that are not images (based on extension). You can add more if needed. image_extensions = [".jpg", ".jpeg", ".png", ".bmp", ".gif", ".tiff"] image_files = [f for f in all_files if any(f.lower().endswith(ext) for ext in image_extensions)] # Open the output file for writing with open(dst_path, 'w') as out_file: for index, img_file in enumerate(image_files): data = { "question_id": index, "image": img_file, "text": vlm_prompt, "category": "detail" } out_file.write(json.dumps(data) + '\n') print(f"Processed {len(image_files)} images. Output saved to {dst_path}.") def generate_answer_id(length=20): # Define the characters that can be used in the string characters = string.ascii_letters + string.digits # Generate a random string of the specified length answer_id = ''.join(random.choice(characters) for _ in range(length)) return answer_id def method(vlm=vlm, llm=llm, ovod=ovod, start_index=None, end_index=None, shard_index=None, num_shards=None): # vlm_question_path = os.path.join(BASE_DIR, 'approach/vlm/sampled_vlm_questions.jsonl') # vlm_question_path = os.path.join(BASE_DIR, f'approach/vlm/icse_test_set_merged_vlm_questions_from671.jsonl') # icse # complete_vlm_question_abl_i_path = os.path.join(BASE_DIR, f'approach/vlm/icse_test_set_merged_complete_vlm_questions_ablation_interactability.jsonl') # icse # complete_vlm_question_path = os.path.join(BASE_DIR, f'approach/vlm/icse_test_set_merged_complete_vlm_questions.jsonl') # vlm_question_path = os.path.join(BASE_DIR, f'approach/vlm/gpt4v_failure_questions.jsonl') # icse_rebuttal # vlm_question_path = os.path.join(BASE_DIR, f'approach/vlm/icse_rbt_{perspective}_vlm_questions.jsonl') # complete_vlm_question_path = os.path.join(BASE_DIR, f'approach/vlm/icse_rbt_{perspective}_complete_vlm_questions.jsonl') # fse # vlm_question_path = os.path.join(BASE_DIR, f'approach/vlm/fse_union3_{vlm}_questions.jsonl') complete_vlm_question_path = os.path.join(BASE_DIR, f'approach/vlm/realfse_union3_complete_{vlm}_questions.jsonl') vlm_question_path = complete_vlm_question_path # vlm_question_path = os.path.join(BASE_DIR, f'approach/vlm/gpt4v_failure_questions.jsonl') # playground/data/coco2014_val_qa_eval/qa90_questions.jsonl # vlm_answer_path = os.path.join(BASE_DIR, f'approach/vlm/{host_device}_icse_test_set_merged_{vlm}_answer_slowerspeed_1.jsonl') # vlm_answer_path = os.path.join(BASE_DIR, f'approach/vlm/{host_device}_icse_test_set_merged_{vlm}_answer_higherspeed_r3_1_from195.jsonl') # icse # vlm_answer_path = os.path.join(BASE_DIR, f'approach/vlm/cuhk_icse_test_set_merged_{vlm}_answer.jsonl') # llava_vlm_answer_path = os.path.join(BASE_DIR, f'approach/vlm/{host_device}_icse_test_set_merged_{vlm}_answer.jsonl') # gemini_vlm_answer_path = os.path.join(BASE_DIR, f'approach/vlm/{host_device}_icse_test_set_merged_{vlm}_answer_ablation_no_interactability_r1_0.jsonl') # gpt4_results_answer_path = os.path.join(BASE_DIR, f'approach/vlm/aaa_cuhk_icse_test_set_merged_gpt4v_answer_higherspeed_r3_0.jsonl') # interactability_abl_results_answer_path = os.path.join(BASE_DIR, f'approach/vlm/aaa_hit_icse_test_set_merged_gemini_answer_ablation_no_interactability_r1_0.jsonl') # icse_rebuttal # vlm_answer_path = os.path.join(BASE_DIR, f'approach/vlm/icse_rbt_{perspective}_{vlm}_answer.jsonl') # llava_vlm_answer_path = os.path.join(BASE_DIR, f'approach/vlm/{host_device}_icse_rbt_{perspective}_{vlm}_answer.jsonl') # gemini_vlm_answer_path = os.path.join(BASE_DIR, f'approach/vlm/{host_device}_icse_rbt_{perspective}_{vlm}_answer.jsonl') # fse vlm_answer_path = os.path.join(BASE_DIR, f'approach/vlm/realfse_uninon3_{vlm}_answer.jsonl') llava_vlm_answer_path = os.path.join(BASE_DIR, f'approach/vlm/{host_device}_fse_uninon3_{vlm}_answer.jsonl') gemini_vlm_answer_path = os.path.join(BASE_DIR, f'approach/vlm/realfse_union3_{vlm}_answer.jsonl') # icse # gpt4_results_answer_path = os.path.join(BASE_DIR, f'approach/vlm/aaa_icse_rbt_{perspective}_gpt4v_answer.jsonl') # icse_rebuttal # gpt4_results_answer_path = os.path.join(BASE_DIR, f'approach/vlm/icse_rbt_{perspective}_{vlm}_answer_265_528.jsonl') # fse gpt4_results_answer_path = os.path.join(BASE_DIR, f'approach/vlm/realfse_uninon3_gpt4v_answer.jsonl') # print(gpt4_results_answer_path) # interactability_abl_results_answer_path = os.path.join(BASE_DIR, f'approach/vlm/aaa_icse_rbt_{perspective}_answer.jsonl') # icse # # /path/to/answer-file-our.jsonl # # llm_candidate_path = os.path.join(BASE_DIR, f'approach/llm/{vlm}_{llm}_cancidate_objects.json') # interactable_object_path = os.path.join(BASE_DIR, f'approach/llm/icse_test_set_merged_{vlm}_{llm}_interactable_objects.json') # ovod_output_dir = os.path.join(BASE_DIR, f'approach/ovod/output/icse_test_set_merged_{ovod}') # # oovd_object_bbox_path = os.path.join(BASE_DIR, f'approach/ovod/icse_test_set_merged_{vlm}_{llm}_{ovod}_object_bbox_gpu3_2.json') # # oovd_object_bbox_path = os.path.join(BASE_DIR, f'approach/ovod/icse_test_set_merged_{vlm}_{llm}_{ovod}_ablation_no_interactability_object_bbox_gpu0.json') # oovd_object_bbox_path = os.path.join(BASE_DIR, f'approach/ovod/icse_test_set_merged_{vlm}_{llm}_{ovod}_ablation_no_feedback_object_bbox_gpu2.json') # icse_rebuttal # interactable_object_path = os.path.join(BASE_DIR, f'approach/llm/icse_rbt_{perspective}_{vlm}_{llm}_interactable_objects.json') # ovod_output_dir = os.path.join(BASE_DIR, f'approach/ovod/output/icse_rbt_{perspective}_{ovod}') # # oovd_object_bbox_path = os.path.join(BASE_DIR, f'approach/ovod/icse_test_set_merged_{vlm}_{llm}_{ovod}_object_bbox_gpu3_2.json') # # oovd_object_bbox_path = os.path.join(BASE_DIR, f'approach/ovod/icse_test_set_merged_{vlm}_{llm}_{ovod}_ablation_no_interactability_object_bbox_gpu0.json') # oovd_object_bbox_path = os.path.join(BASE_DIR, f'approach/ovod/icse_rbt_{perspective}_{vlm}_{llm}_{ovod}_object_bbox_gpu3_265_528.json') gpu = '3' # fse interactable_object_path = os.path.join(BASE_DIR, f'approach/llm/realfse_union3_{vlm}_{llm}_interactable_objects.json') # TODO: Tentatively unused ovod_output_dir = os.path.join(BASE_DIR, f'approach/ovod/output/realfse_union3_{ovod}') # GPU 0 oovd_object_bbox_path = os.path.join(BASE_DIR, f'approach/ovod/realfse_union3_{vlm}_{ovod}_object_bbox_gpu{gpu}.json') oovd_object_bbox_path = output_path_for_selection( oovd_object_bbox_path, start_index=start_index, end_index=end_index, shard_index=shard_index, num_shards=num_shards, ) llava7b_model_path = os.path.join(BASE_DIR, 'approach/vlm/LLaVA/checkpoints/llava-v1.5-7b') llava7b_cmd = f''' python model_vqa.py \ --model-path {llava7b_model_path} \ --question-file \ {complete_vlm_question_path} \ --image-folder \ {images_dir} \ --answers-file \ {llava_vlm_answer_path} ''' # STEP #0 # First time of running # TODO: Check whether it is first-time running # generate_question_file(images_dir, vlm_question_path) # STEP #1 # VLM - Get local context # print(f'VLM {vlm} analysis begins ...') # vlm_start_time = time.time() # logging.info(f'STEP #1 VLM {vlm} analysis started at %s', time.strftime('%Y-%m-%d %H:%M:%S', time.localtime())) # if vlm == 'llava7b': # original_path = os.path.dirname(__file__) # os.chdir(llava_path) # os.system(llava7b_cmd) # os.chdir(original_path) # elif vlm == 'generate_q_file': # from approach.vlm.gpt4v.gpt4v import process_image_q # with open(vlm_question_path, 'r') as q_file, open(complete_vlm_question_path, 'w') as cq_file: # Open answer file in append mode # qfile_lines = q_file.readlines() # key_idx = 0 # for line in tqdm(qfile_lines): # key_idx = (key_idx + 1) % 4 # ans_item = {} # line_data = json.loads(line) # image_path = os.path.join(images_dir, line_data['image']) # image_question = line_data['text'] # if vlm == 'gpt4v_abl': # gpt4v_ablation = True # elif vlm == 'gpt4v': # gpt4v_ablation = False # gpt4v_q = process_image_q(image_question, image_path, key_idx) # data = { # "question_id": line_data['question_id'], # "image": line_data['image'], # "text": gpt4v_q, # "category": "detail" # } # cq_file.write(json.dumps(data) + '\n') # elif vlm == 'gpt4v' or vlm == 'gpt4v_abl' or vlm == 'claude35sonnet' or vlm == 'gemini15pro': # from approach.vlm.gpt4v.gpt4v import process_image # with open(vlm_question_path, 'r') as q_file, open(vlm_answer_path, 'a') as a_file: # Open answer file in append mode # qfile_lines = q_file.readlines() # key_idx = 0 # for line in tqdm(qfile_lines): # key_idx = (key_idx + 1) % 4 # ans_item = {} # line_data = json.loads(line) # image_path = os.path.join(images_dir, line_data['image']) # image_question = line_data['text'] # gpt4v_ablation = False # if vlm == 'gpt4v_abl': # gpt4v_ablation = True # elif vlm == 'gpt4v': # gpt4v_ablation = False # gpt4v_res = process_image(vlm, image_question, image_path, gpt4v_ablation, key_idx) # print(gpt4v_res) # # time.sleep(3) # ans_item = { # "question_id": line_data['question_id'], # "prompt": '', # "text": gpt4v_res, # "answer_id": generate_answer_id(), # "model_id": vlm, # "metadata": {} # } # a_file.write(json.dumps(ans_item) + '\n') # a_file.flush() # elif vlm == 'bing': # from approach.vlm.bing.bing import context_conversation # with open(vlm_question_path, 'r') as q_file, open(vlm_answer_path, 'a') as a_file: # Open answer file in append mode # for line in tqdm(q_file): # ans_item = {} # line_data = json.loads(line) # image_path = os.path.join(images_dir, line_data['image']) # image_question = line_data['text'] # bing_res = context_conversation(image_question, image_path) # print(bing_res) # time.sleep(15) # ans_item = { # "question_id": line_data['question_id'], # "prompt": image_question, # "text": bing_res, # "answer_id": generate_answer_id(), # "model_id": vlm, # "metadata": {} # } # a_file.write(json.dumps(ans_item) + '\n') # elif vlm == 'gemini': # import pathlib # import textwrap # import google.generativeai as genai # GOOGLE_API_KEY = os.environ.get("GOOGLE_API_KEY", "[GOOGLE_API_KEY]") # genai.configure(api_key=GOOGLE_API_KEY) # safety_settings = [ # { # "category": "HARM_CATEGORY_DANGEROUS", # "threshold": "BLOCK_NONE", # }, # { # "category": "HARM_CATEGORY_HARASSMENT", # "threshold": "BLOCK_NONE", # }, # { # "category": "HARM_CATEGORY_HATE_SPEECH", # "threshold": "BLOCK_NONE", # }, # { # "category": "HARM_CATEGORY_SEXUALLY_EXPLICIT", # "threshold": "BLOCK_NONE", # }, # { # "category": "HARM_CATEGORY_DANGEROUS_CONTENT", # "threshold": "BLOCK_NONE", # }, # ] # # before fse # # model = genai.GenerativeModel('gemini-pro-vision') # # fse # model = genai.GenerativeModel('gemini-1.5-pro') # with open(vlm_question_path, 'r') as q_file, open(gemini_vlm_answer_path, 'a') as a_file: # Open answer file in append mode # qfile_lines = q_file.readlines()[0:2] # for line in tqdm(qfile_lines): # ans_item = {} # line_data = json.loads(line) # image_path = os.path.join(images_dir, line_data['image']) # image_question = line_data['text'] # gemini_response = model.generate_content([image_question, PIL.Image.open(image_path)], safety_settings=safety_settings) # gemini_response.resolve() # try: # gemini_response = gemini_response.text # ans_item = { # "question_id": line_data['question_id'], # "prompt": image_question, # "text": gemini_response, # "answer_id": generate_answer_id(), # "model_id": vlm, # "metadata": {} # } # a_file.write(json.dumps(ans_item) + '\n') # a_file.flush() # except Exception as e: # print(f"Error for image {image_path}. Error: {e}") # # print(gemini_response) # time.sleep(30) # elif vlm.endswith('_pass'): # print(f'Passing VLM {vlm} ...') # else: # raise Exception('Unrecognized VLM!') # logging.info(f'STEP #1 VLM {vlm} analysis completed at %s', time.strftime('%Y-%m-%d %H:%M:%S', time.localtime())) # logging.info(f'STEP #1 VLM {vlm} analysis total time taken: {time.time() - vlm_start_time} seconds') # Get label candidates # if llm == 'gpt_3.5_turbo': # from approach.llm.gpt_polling import infer_object_candidates # elif llm == 'llama2': # from approach.llm.llama import infer_object_candidates # STEP #3 # if llm == 'gpt_3.5_turbo' or llm == 'gpt_3.5_turbo_abl': # from approach.llm.gpt_polling import infer_objects # if llm == 'gpt_3.5_turbo_abl': # gpt35_ablation = True # elif llm == 'gpt_3.5_turbo': # gpt35_ablation = False # infer_objects(vlm_question_path, vlm_answer_path, interactable_object_path, gpt35_ablation) # elif llm == 'llama2': # from approach.llm.llama import infer_objects # exit(0) # elif llm.endswith('_pass'): # print(f'Passing LLM {llm} ...') # else: # raise Exception('Unrecognized LLM!') # STEP #2 # Open-vocabulary object detection ovod_start_time = time.time() logging.info(f'STEP #2 OVOD {ovod} analysis started at %s', time.strftime('%Y-%m-%d %H:%M:%S', time.localtime())) vlm_questions = {} with open(complete_vlm_question_path, 'r') as q_file: for line in q_file: line_data = json.loads(line) vlm_questions[line_data['question_id']] = line_data # print(line_data['question_id']) if ovod == 'grounding_dino': from approach.ovod.GroundingDINO.demo.inference_on_a_image import process_grounding_dino original_path = os.path.dirname(__file__) os.chdir(grounding_dino_path) # with open(llm_candidate_path, 'r') as llm_candidate_file, open(oovd_object_bbox_path, 'w') as oovd_file: with open(interactable_object_path, 'r') as llm_candidate_file, open(oovd_object_bbox_path, 'w') as oovd_file: llm_candidate = json.load(llm_candidate_file) all_oovd_res = {} for image_index in llm_candidate.keys(): print(image_index) # TODO: if (image_index != '1160'): ovod_image_path = os.path.join(images_dir, vlm_questions[int(image_index)]["image"]) ovod_candidates = llm_candidate[image_index]['interactable_objects'] # grounding_dino_command = generate_grounding_dino_command(ovod_image_path, ovod_output_dir, ovod_candidates) # print(grounding_dino_command) # os.system(grounding_dino_command) gdino_res = process_grounding_dino( config_file='groundingdino/config/GroundingDINO_SwinT_OGC.py', checkpoint_path='weights/groundingdino_swint_ogc.pth', image_path=ovod_image_path, ovod_candidates=ovod_candidates, output_dir=ovod_output_dir, box_threshold=0.3, text_threshold=0.25, token_spans=None, cpu_only=False ) # print(gdino_res) object_oovd_item = copy.deepcopy(llm_candidate[image_index]) object_oovd_item['oovd_result'] = gdino_res print(object_oovd_item) all_oovd_res[image_index] = object_oovd_item json.dump(all_oovd_res, oovd_file, indent=4) os.chdir(original_path) elif ovod == 'ape_d' or ovod == 'ape_d_abl': from approach.ovod.APE.demo.ape_inference import run_ape_model_inference original_path = os.path.dirname(__file__) os.chdir(ape_path) # with open(llm_candidate_path, 'r') as llm_candidate_file, open(oovd_object_bbox_path, 'w') as oovd_file: all_ape_res = [] # General with open(vlm_answer_path, 'r') as llm_candidate_file, open(oovd_object_bbox_path, 'w') as oovd_file: # GPT-4 eval # with open(gpt4_results_answer_path, 'r') as llm_candidate_file, open(oovd_object_bbox_path, 'w') as oovd_file: # Gemini abl i eval # with open(interactability_abl_results_answer_path, 'r') as llm_candidate_file, open(oovd_object_bbox_path, 'w') as oovd_file: # llm_candidate = json.load(llm_candidate_file) all_oovd_res = {} # candidate_lines = llm_candidate_file.readlines()[265:528] # GPU 0 # candidate_lines = llm_candidate_file.readlines()[0:230] # candidate_lines = llm_candidate_file.readlines()[230:460] # candidate_lines = llm_candidate_file.readlines()[460:690] candidate_lines = select_jsonl_lines( llm_candidate_file.readlines(), start_index=start_index, end_index=end_index, shard_index=shard_index, num_shards=num_shards, ) # candidate_lines = llm_candidate_file.readlines() for line in tqdm(candidate_lines): ans_item = {} line_data = json.loads(line) # for image_index in llm_candidate.keys(): # print(image_index) # TODO: # if (image_index != '1160'): # image_index = line_data['question_id'] image_name = vlm_questions[line_data['question_id']]["image"] ovod_image_path = os.path.join(images_dir, image_name) # General # try: # if line_data['text'].startswith(" ```json"): # ovod_candidates = json.loads(line_data['text'][8:-4])['objects'] # elif line_data['text'].startswith(' {\"objects\"'): # ovod_candidates = json.loads(line_data['text'])['objects'] # else: # print(f"Error for decoding IVO json for {image_name}.") # except Exception as e: # print(f"Error for decoding IVO json for image {image_name}. Error: {e}") # continue # GPT-4v if line_data['text']: ovod_candidates = line_data['text']['objects'] else: continue all_res = [] for ocd in ovod_candidates.keys(): referring_expr_str = ovod_candidates[ocd] translator = str.maketrans('', '', string.punctuation) if type(referring_expr_str) is str: referring_expr_str = referring_expr_str.translate(translator) else: # dict referring_expr_str = ' '.join(referring_expr_str.values()) referring_expr_str = referring_expr_str.translate(translator) referring_expr_str = f'{ocd}: {referring_expr_str}' # print(image_name, referring_expr_str) all_res.append(referring_expr_str) if ovod == 'ape_d': ape_threshold = 0.15 elif ovod == 'ape_d_abl': ape_threshold = 0.1 ape_res = [] try: ape_res = run_ape_model_inference( config_file='configs/LVISCOCOCOCOSTUFF_O365_OID_VGR_SA1B_REFCOCO_GQA_PhraseCut_Flickr30k/ape_deta/ape_deta_vitl_eva02_clip_vlf_lsj1024_cp_16x4_1080k.py', input_path=ovod_image_path, # GPU 0 output_path=f'./realfse_{vlm}_{ovod}_gpu0123', confidence_threshold=ape_threshold, text_prompt=', '.join(all_res), with_box=True, with_mask=False, with_sseg=False, opts=[ "train.init_checkpoint='./ape_d_model_final.pth'", "model.model_language.cache_dir=''", "model.model_vision.select_box_nums_for_evaluation=500", "model.model_vision.text_feature_bank_reset=True", "model.model_vision.backbone.net.xattn=False", "model.model_vision.transformer.encoder.pytorch_attn=True", "model.model_vision.transformer.decoder.pytorch_attn=True" ] ) except Exception as e: print(f"Error for image {image_name}. Error: {e}") all_ape_res.extend(enrich_ape_results(ape_res, image_name, extract_image_id)) # object_oovd_item = copy.deepcopy(llm_candidate[image_index]) # object_oovd_item['ape_result'] = all_ape_res # # print(object_oovd_item) # all_oovd_res[image_index] = object_oovd_item write_json_atomic(oovd_object_bbox_path, all_ape_res) os.chdir(original_path) elif ovod.endswith('_pass'): print(f'Passing OVOD {ovod} ...') else: raise Exception('Unrecognized OVOD!') logging.info(f'STEP #2 OVOD {ovod} analysis completed at %s', time.strftime('%Y-%m-%d %H:%M:%S', time.localtime())) logging.info(f'STEP #2 OVOD {ovod} analysis total time taken: {time.time() - ovod_start_time} seconds') # icse & fse def extract_image_id(image_name): # Split the image name into parts and form the image_id accordingly parts = image_name.split('_') print(image_name) base, extension = parts[1].split('.') return int(parts[0] + base.zfill(3)) # icse_rebuttal # def extract_image_id(image_name): # # Split the image name into parts by underscore # parts = image_name.split('_') # # Extracting the first part as the base and the numeric portion of the third part before the file extension # base = parts[0] # This will give '625470' # numeric_part = parts[2].split('.')[0] # This will give 'b3' # # Removing non-numeric characters from 'b3' # numeric_part = ''.join(filter(str.isdigit, numeric_part)) # # Zfill is used to ensure the numeric part has at least 3 digits, then combining with base # print(int(base + numeric_part.zfill(3))) # return int(base + numeric_part.zfill(3)) def main(): vlms = ['llava7b', 'bing', 'gpt4v'] llms = ['llama2', 'gpt_3.5_turbo'] ovods = ['grounding_dino', 'glip', 'ape_d'] # python method.py -v gpt4v -l gpt_3.5_turbo -o ape_d # python method.py -v gpt4v -l gpt_3.5_turbo -o ape_d_abl # python method.py -v claude35sonnet -l gpt_3.5_turbo -o ape_d # python method.py -v gemini -l gpt_3.5_turbo -o ape_d if __name__=='__main__': # main() parser = argparse.ArgumentParser("Method", add_help=True) parser.add_argument("--vlm", "-v", type=str, required=True, help="vlm") parser.add_argument("--llm", "-l", type=str, required=True, help="llm") parser.add_argument("--ovod", "-o", type=str, required=True, help="ovod") parser.add_argument("--start-index", type=int, default=None, help="first JSONL row to process") parser.add_argument("--end-index", type=int, default=None, help="exclusive JSONL row end") parser.add_argument("--shard-index", type=int, default=None, help="zero-based shard index") parser.add_argument("--num-shards", type=int, default=None, help="total number of shards") args = parser.parse_args() gpu = '3' stage = '2' idx = '0001' logging.basicConfig(format='%(asctime)s - %(levelname)s - %(name)s - %(message)s', datefmt='%m/%d/%Y %H:%M:%S', level=logging.INFO, # gpu filename=f'../log/realfse/240910_{args.vlm}_{stage}_g{gpu}_{idx}.log' ) method( args.vlm, args.llm, args.ovod, start_index=args.start_index, end_index=args.end_index, shard_index=args.shard_index, num_shards=args.num_shards, ) # CUDA_VISIBLE_DEVICES=0 python method.py -v gpt4v -l gpt_3.5_turbo -o ape_d > ../log/realfse/240910_gpt4v_2_g0_0001.txt # CUDA_VISIBLE_DEVICES=3 python method_gemini.py -v gemini15pro -l gpt_3.5_turbo -o ape_d > ../log/realfse/240910_gemini15pro_2_g3_0001.txt