diff --git a/env_files/minicpm_v_4d5_requirements.txt b/env_files/minicpm_v_4d5_requirements.txt new file mode 100644 index 00000000..f2935321 --- /dev/null +++ b/env_files/minicpm_v_4d5_requirements.txt @@ -0,0 +1,13 @@ +torch==2.4.0 +torchvision==0.19.0 +transformers==4.55.0 +accelerate +datasets +pandas +numpy==1.26.4 +Pillow +sentencepiece +protobuf +einops +timm +decord diff --git a/mmeval/infer/minicpm_v_4d5.py b/mmeval/infer/minicpm_v_4d5.py new file mode 100644 index 00000000..7b5b8e8d --- /dev/null +++ b/mmeval/infer/minicpm_v_4d5.py @@ -0,0 +1,93 @@ +"""minicpm_v_4d5 — MiniCPM-V-4_5. + +HF: https://huggingface.co/openbmb/MiniCPM-V-4_5 +GH: https://github.com/OpenBMB/MiniCPM-o +""" +import re +import copy + +import torch +from PIL import Image +from transformers import AutoModel, AutoTokenizer + +from mmeval.infer.task import Task +from mmeval.utils import constants +from mmeval.utils.argparser import parse_args, parse_model_kwargs, parse_gen_kwargs + + +class TaskRunner(Task): + def __init__(self, args): + self.args = args + self.device = torch.device("cuda" if torch.cuda.is_available() else "cpu") + self.dtype = getattr(args, "dtype") or torch.bfloat16 + self.default_model_kwargs = { + "attn_implementation": "sdpa", + } + self.default_gen_kwargs = {"max_new_tokens": 1024} + self.model_kwargs = parse_model_kwargs(args, self.default_model_kwargs) + self.gen_kwargs = parse_gen_kwargs(args, self.default_gen_kwargs) + + super().__init__(args) + + def load_model(self, args): + self.model = AutoModel.from_pretrained( + args.model_name_or_path, + torch_dtype=self.dtype, + trust_remote_code=True, + **self.model_kwargs, + ).eval().cuda() + self.tokenizer = AutoTokenizer.from_pretrained( + args.model_name_or_path, trust_remote_code=True, + ) + + def parse_input(self, message): + question = message["prompt"] + q_chunks = re.split(r'(<(?:image|video)>)', question) + media_list = message.get('media', []) + + content = [] + text_parts = [] + media_idx = 0 + for chunk in q_chunks: + if not chunk.strip(): + continue + if chunk == constants.image: + img = media_list[media_idx] + if isinstance(img, str): + img = Image.open(img).convert("RGB") + elif hasattr(img, "convert"): + img = img.convert("RGB") + content.append(img) + media_idx += 1 + elif chunk == constants.video: + raise NotImplementedError("minicpm_v_4d5 video input not implemented") + else: + text_parts.append(chunk) + content.append("".join(text_parts).strip()) + return [{"role": "user", "content": content}] + + def run_sample(self, sample: dict): + if self.args.score_target: + raise NotImplementedError( + "minicpm_v_4d5: score_target is not implemented yet" + ) + + ori_sample = copy.deepcopy(sample) + message = sample["messages"][0] + msgs = self.parse_input(message) + + response = self.model.chat( + msgs=msgs, + tokenizer=self.tokenizer, + enable_thinking=False, + stream=False, + **self.gen_kwargs, + ) + ori_sample["messages"].append({"role": "assistant", "response": response}) + return ori_sample + + +if __name__ == "__main__": + args = parse_args() + model_evaluator = TaskRunner(args) + model_evaluator.inference_dataset() diff --git a/mmeval/registry.py b/mmeval/registry.py index dfbf0b98..be8e4109 100644 --- a/mmeval/registry.py +++ b/mmeval/registry.py @@ -77,6 +77,7 @@ "doubao-seed-2-0-mini-260215", "doubao-seed-2-0-lite-260215", "doubao-seed-2-0-code-preview-260215", "doubao-seed-2-0-pro-260215"], "hunyuan_vision": ["hunyuan-vision", "hunyuan-vision-1.5-instruct", "hunyuan-t1-vision", "hunyuan-turbos-vision", "hunyuan-large-vision"], "cosmos_reason2": ["Cosmos-Reason2-2B", "Cosmos-Reason2-8B"], + "minicpm_v_4d5": ["MiniCPM-V-4_5"], } series_infer_env_mapping = { @@ -316,4 +317,8 @@ "env": os.path.join(env_dir, "cosmos_reason2"), "infer_file": "cosmos_reason2.py", }, + "minicpm_v_4d5": { + "env": os.path.join(env_dir, "minicpm_v_4d5"), + "infer_file": "minicpm_v_4d5.py", + }, } diff --git a/test_results/minicpm_v_4d5_2026-06-10/no_media/result.json b/test_results/minicpm_v_4d5_2026-06-10/no_media/result.json new file mode 100644 index 00000000..19b29031 --- /dev/null +++ b/test_results/minicpm_v_4d5_2026-06-10/no_media/result.json @@ -0,0 +1,44 @@ +[ + { + "id": 0, + "media": [], + "messages": [ + { + "role": "user", + "question": "What is the result of 1 plus 1?", + "answer": "", + "options": {}, + "choices": [], + "prompt": "What is the result of 1 plus 1?", + "hint": "" + }, + { + "role": "assistant", + "response": "The result of 1 plus 1 is 2." + } + ], + "comment": "Test 'no media' modality.", + "eval-id": 0 + }, + { + "id": 1, + "media": [], + "messages": [ + { + "role": "user", + "question": "What is the capital city of China?", + "answer": "", + "options": {}, + "choices": [], + "prompt": "What is the capital city of China?", + "hint": "" + }, + { + "role": "assistant", + "response": "The capital city of China is Beijing." + } + ], + "comment": "Test 'no media' modality.", + "eval-id": 1 + } +] \ No newline at end of file diff --git a/test_results/minicpm_v_4d5_2026-06-10/single_image_start/result.json b/test_results/minicpm_v_4d5_2026-06-10/single_image_start/result.json new file mode 100644 index 00000000..1dfd25b8 --- /dev/null +++ b/test_results/minicpm_v_4d5_2026-06-10/single_image_start/result.json @@ -0,0 +1,48 @@ +[ + { + "id": 0, + "media": [ + "truck.png" + ], + "messages": [ + { + "role": "user", + "question": " Please provide a detailed description of the contents shown in the image.", + "answer": "", + "options": {}, + "choices": [], + "prompt": " Please provide a detailed description of the contents shown in the image.", + "hint": "" + }, + { + "role": "assistant", + "response": "The image features a GMC pickup truck prominently displayed in an outdoor setting, likely on sandy or dusty terrain. The vehicle is captured from a front three-quarter angle, showcasing its bold and robust design. It has a dark metallic gray exterior finish which gives it a sleek yet rugged appearance.\n\nThe front fascia of the truck includes" + } + ], + "comment": "Test 'single image start' modality.", + "eval-id": 0 + }, + { + "id": 1, + "media": [ + "boat.png" + ], + "messages": [ + { + "role": "user", + "question": " Please provide a detailed description of the contents shown in the image.", + "answer": "", + "options": {}, + "choices": [], + "prompt": " Please provide a detailed description of the contents shown in the image.", + "hint": "" + }, + { + "role": "assistant", + "response": "The image depicts a sleek, modern yacht cruising at high speed across the ocean. The vessel is predominantly white with clean lines and an aerodynamic design, emphasizing luxury and performance. It features large windows along the side of the cabin area, suggesting spacious interiors likely equipped with advanced amenities for comfort and entertainment.\n\nThe yacht has two" + } + ], + "comment": "Test 'single image start' modality.", + "eval-id": 1 + } +] \ No newline at end of file diff --git a/test_results/minicpm_v_4d5_2026-06-10/test_summary.json b/test_results/minicpm_v_4d5_2026-06-10/test_summary.json new file mode 100644 index 00000000..63b128ea --- /dev/null +++ b/test_results/minicpm_v_4d5_2026-06-10/test_summary.json @@ -0,0 +1,20 @@ +{ + "model_series": "minicpm_v_4d5", + "model_name": "openbmb/MiniCPM-V-4_5", + "test_date": "2026-06-10", + "conda_env": "/raid/ztw/envs/minicpm_v_4d5", + "env_summary": "transformers==4.55.0", + "status": "PASS", + "modalities_tested": { + "no_media": { + "status": "pass", + "rows": 2, + "result_path": "/raid/ztw/simple-mmeval-test-result/work_dirs/minicpm_v_4d5/2026-06-10/no_media/result.json" + }, + "single_image_start": { + "status": "pass", + "rows": 2, + "result_path": "/raid/ztw/simple-mmeval-test-result/work_dirs/minicpm_v_4d5/2026-06-10/single_image_start/result.json" + } + } +} \ No newline at end of file diff --git a/test_results/test_minicpm_v_4d5.sh b/test_results/test_minicpm_v_4d5.sh new file mode 100755 index 00000000..92d402ed --- /dev/null +++ b/test_results/test_minicpm_v_4d5.sh @@ -0,0 +1,25 @@ +#!/usr/bin/env bash +# Smoke test for minicpm_v_4d5 (model: openbmb/MiniCPM-V-4_5). +set -euo pipefail + +REPO_DIR=${REPO_DIR:-/raid/ztw/simple-mmeval-skills-dev} +OUT_ROOT=${OUT_ROOT:-/raid/ztw/simple-mmeval-test-result/work_dirs/minicpm_v_4d5/$(date +%Y-%m-%d)} +GPU=${GPU:-0} +MODEL_ID=openbmb/MiniCPM-V-4_5 + +cd "$REPO_DIR" +mkdir -p "$OUT_ROOT" + +for spec in "no_media|32|" "single_image_start|64|tests/media/448"; do + IFS='|' read -r sample maxtok imgdir <<< "$spec" + out="$OUT_ROOT/$sample" + extra=() + [[ -n "$imgdir" ]] && extra=(--img_dir "$imgdir") + PYTHONPATH="$REPO_DIR" ENV_DIR=/raid/ztw/envs CUDA_VISIBLE_DEVICES="$GPU" \ + /raid/ztw/envs/minicpm_v_4d5/bin/python mmeval/run.py \ + --model_name_or_path "$MODEL_ID" \ + --dataset local@json --infile tests/samples/$sample.json \ + --out_dir "$out" \ + --gpu_per_parallel 1 --parallel_per_task 1 --max_new_tokens "$maxtok" \ + "${extra[@]}" +done