From 5ec262b58ca6254256c3b6921f08b0d4e1a71b4f Mon Sep 17 00:00:00 2001 From: ZTWHHH Date: Wed, 10 Jun 2026 10:05:30 +0000 Subject: [PATCH] Add minicpm_v_4d5 model integration (openbmb/MiniCPM-V-4_5) - mmeval/infer/minicpm_v_4d5.py: official-style loading and decoding - env_files/minicpm_v_4d5_requirements.txt: pinned reproducible env - mmeval/registry.py: series_mapping + series_infer_env_mapping entries - test_results/: passing no_media + single_image_start smoke runs HF: https://huggingface.co/openbmb/MiniCPM-V-4_5 GH: https://github.com/OpenBMB/MiniCPM-o Co-Authored-By: Claude Opus 4.7 --- env_files/minicpm_v_4d5_requirements.txt | 13 +++ mmeval/infer/minicpm_v_4d5.py | 93 +++++++++++++++++++ mmeval/registry.py | 5 + .../no_media/result.json | 44 +++++++++ .../single_image_start/result.json | 48 ++++++++++ .../test_summary.json | 20 ++++ test_results/test_minicpm_v_4d5.sh | 25 +++++ 7 files changed, 248 insertions(+) create mode 100644 env_files/minicpm_v_4d5_requirements.txt create mode 100644 mmeval/infer/minicpm_v_4d5.py create mode 100644 test_results/minicpm_v_4d5_2026-06-10/no_media/result.json create mode 100644 test_results/minicpm_v_4d5_2026-06-10/single_image_start/result.json create mode 100644 test_results/minicpm_v_4d5_2026-06-10/test_summary.json create mode 100755 test_results/test_minicpm_v_4d5.sh diff --git a/env_files/minicpm_v_4d5_requirements.txt b/env_files/minicpm_v_4d5_requirements.txt new file mode 100644 index 00000000..f2935321 --- /dev/null +++ b/env_files/minicpm_v_4d5_requirements.txt @@ -0,0 +1,13 @@ +torch==2.4.0 +torchvision==0.19.0 +transformers==4.55.0 +accelerate +datasets +pandas +numpy==1.26.4 +Pillow +sentencepiece +protobuf +einops +timm +decord diff --git a/mmeval/infer/minicpm_v_4d5.py b/mmeval/infer/minicpm_v_4d5.py new file mode 100644 index 00000000..7b5b8e8d --- /dev/null +++ b/mmeval/infer/minicpm_v_4d5.py @@ -0,0 +1,93 @@ +"""minicpm_v_4d5 — MiniCPM-V-4_5. + +HF: https://huggingface.co/openbmb/MiniCPM-V-4_5 +GH: https://github.com/OpenBMB/MiniCPM-o +""" +import re +import copy + +import torch +from PIL import Image +from transformers import AutoModel, AutoTokenizer + +from mmeval.infer.task import Task +from mmeval.utils import constants +from mmeval.utils.argparser import parse_args, parse_model_kwargs, parse_gen_kwargs + + +class TaskRunner(Task): + def __init__(self, args): + self.args = args + self.device = torch.device("cuda" if torch.cuda.is_available() else "cpu") + self.dtype = getattr(args, "dtype") or torch.bfloat16 + self.default_model_kwargs = { + "attn_implementation": "sdpa", + } + self.default_gen_kwargs = {"max_new_tokens": 1024} + self.model_kwargs = parse_model_kwargs(args, self.default_model_kwargs) + self.gen_kwargs = parse_gen_kwargs(args, self.default_gen_kwargs) + + super().__init__(args) + + def load_model(self, args): + self.model = AutoModel.from_pretrained( + args.model_name_or_path, + torch_dtype=self.dtype, + trust_remote_code=True, + **self.model_kwargs, + ).eval().cuda() + self.tokenizer = AutoTokenizer.from_pretrained( + args.model_name_or_path, trust_remote_code=True, + ) + + def parse_input(self, message): + question = message["prompt"] + q_chunks = re.split(r'(<(?:image|video)>)', question) + media_list = message.get('media', []) + + content = [] + text_parts = [] + media_idx = 0 + for chunk in q_chunks: + if not chunk.strip(): + continue + if chunk == constants.image: + img = media_list[media_idx] + if isinstance(img, str): + img = Image.open(img).convert("RGB") + elif hasattr(img, "convert"): + img = img.convert("RGB") + content.append(img) + media_idx += 1 + elif chunk == constants.video: + raise NotImplementedError("minicpm_v_4d5 video input not implemented") + else: + text_parts.append(chunk) + content.append("".join(text_parts).strip()) + return [{"role": "user", "content": content}] + + def run_sample(self, sample: dict): + if self.args.score_target: + raise NotImplementedError( + "minicpm_v_4d5: score_target is not implemented yet" + ) + + ori_sample = copy.deepcopy(sample) + message = sample["messages"][0] + msgs = self.parse_input(message) + + response = self.model.chat( + msgs=msgs, + tokenizer=self.tokenizer, + enable_thinking=False, + stream=False, + **self.gen_kwargs, + ) + ori_sample["messages"].append({"role": "assistant", "response": response}) + return ori_sample + + +if __name__ == "__main__": + args = parse_args() + model_evaluator = TaskRunner(args) + model_evaluator.inference_dataset() diff --git a/mmeval/registry.py b/mmeval/registry.py index dfbf0b98..be8e4109 100644 --- a/mmeval/registry.py +++ b/mmeval/registry.py @@ -77,6 +77,7 @@ "doubao-seed-2-0-mini-260215", "doubao-seed-2-0-lite-260215", "doubao-seed-2-0-code-preview-260215", "doubao-seed-2-0-pro-260215"], "hunyuan_vision": ["hunyuan-vision", "hunyuan-vision-1.5-instruct", "hunyuan-t1-vision", "hunyuan-turbos-vision", "hunyuan-large-vision"], "cosmos_reason2": ["Cosmos-Reason2-2B", "Cosmos-Reason2-8B"], + "minicpm_v_4d5": ["MiniCPM-V-4_5"], } series_infer_env_mapping = { @@ -316,4 +317,8 @@ "env": os.path.join(env_dir, "cosmos_reason2"), "infer_file": "cosmos_reason2.py", }, + "minicpm_v_4d5": { + "env": os.path.join(env_dir, "minicpm_v_4d5"), + "infer_file": "minicpm_v_4d5.py", + }, } diff --git a/test_results/minicpm_v_4d5_2026-06-10/no_media/result.json b/test_results/minicpm_v_4d5_2026-06-10/no_media/result.json new file mode 100644 index 00000000..19b29031 --- /dev/null +++ b/test_results/minicpm_v_4d5_2026-06-10/no_media/result.json @@ -0,0 +1,44 @@ +[ + { + "id": 0, + "media": [], + "messages": [ + { + "role": "user", + "question": "What is the result of 1 plus 1?", + "answer": "", + "options": {}, + "choices": [], + "prompt": "What is the result of 1 plus 1?", + "hint": "" + }, + { + "role": "assistant", + "response": "The result of 1 plus 1 is 2." + } + ], + "comment": "Test 'no media' modality.", + "eval-id": 0 + }, + { + "id": 1, + "media": [], + "messages": [ + { + "role": "user", + "question": "What is the capital city of China?", + "answer": "", + "options": {}, + "choices": [], + "prompt": "What is the capital city of China?", + "hint": "" + }, + { + "role": "assistant", + "response": "The capital city of China is Beijing." + } + ], + "comment": "Test 'no media' modality.", + "eval-id": 1 + } +] \ No newline at end of file diff --git a/test_results/minicpm_v_4d5_2026-06-10/single_image_start/result.json b/test_results/minicpm_v_4d5_2026-06-10/single_image_start/result.json new file mode 100644 index 00000000..1dfd25b8 --- /dev/null +++ b/test_results/minicpm_v_4d5_2026-06-10/single_image_start/result.json @@ -0,0 +1,48 @@ +[ + { + "id": 0, + "media": [ + "truck.png" + ], + "messages": [ + { + "role": "user", + "question": " Please provide a detailed description of the contents shown in the image.", + "answer": "", + "options": {}, + "choices": [], + "prompt": " Please provide a detailed description of the contents shown in the image.", + "hint": "" + }, + { + "role": "assistant", + "response": "The image features a GMC pickup truck prominently displayed in an outdoor setting, likely on sandy or dusty terrain. The vehicle is captured from a front three-quarter angle, showcasing its bold and robust design. It has a dark metallic gray exterior finish which gives it a sleek yet rugged appearance.\n\nThe front fascia of the truck includes" + } + ], + "comment": "Test 'single image start' modality.", + "eval-id": 0 + }, + { + "id": 1, + "media": [ + "boat.png" + ], + "messages": [ + { + "role": "user", + "question": " Please provide a detailed description of the contents shown in the image.", + "answer": "", + "options": {}, + "choices": [], + "prompt": " Please provide a detailed description of the contents shown in the image.", + "hint": "" + }, + { + "role": "assistant", + "response": "The image depicts a sleek, modern yacht cruising at high speed across the ocean. The vessel is predominantly white with clean lines and an aerodynamic design, emphasizing luxury and performance. It features large windows along the side of the cabin area, suggesting spacious interiors likely equipped with advanced amenities for comfort and entertainment.\n\nThe yacht has two" + } + ], + "comment": "Test 'single image start' modality.", + "eval-id": 1 + } +] \ No newline at end of file diff --git a/test_results/minicpm_v_4d5_2026-06-10/test_summary.json b/test_results/minicpm_v_4d5_2026-06-10/test_summary.json new file mode 100644 index 00000000..63b128ea --- /dev/null +++ b/test_results/minicpm_v_4d5_2026-06-10/test_summary.json @@ -0,0 +1,20 @@ +{ + "model_series": "minicpm_v_4d5", + "model_name": "openbmb/MiniCPM-V-4_5", + "test_date": "2026-06-10", + "conda_env": "/raid/ztw/envs/minicpm_v_4d5", + "env_summary": "transformers==4.55.0", + "status": "PASS", + "modalities_tested": { + "no_media": { + "status": "pass", + "rows": 2, + "result_path": "/raid/ztw/simple-mmeval-test-result/work_dirs/minicpm_v_4d5/2026-06-10/no_media/result.json" + }, + "single_image_start": { + "status": "pass", + "rows": 2, + "result_path": "/raid/ztw/simple-mmeval-test-result/work_dirs/minicpm_v_4d5/2026-06-10/single_image_start/result.json" + } + } +} \ No newline at end of file diff --git a/test_results/test_minicpm_v_4d5.sh b/test_results/test_minicpm_v_4d5.sh new file mode 100755 index 00000000..92d402ed --- /dev/null +++ b/test_results/test_minicpm_v_4d5.sh @@ -0,0 +1,25 @@ +#!/usr/bin/env bash +# Smoke test for minicpm_v_4d5 (model: openbmb/MiniCPM-V-4_5). +set -euo pipefail + +REPO_DIR=${REPO_DIR:-/raid/ztw/simple-mmeval-skills-dev} +OUT_ROOT=${OUT_ROOT:-/raid/ztw/simple-mmeval-test-result/work_dirs/minicpm_v_4d5/$(date +%Y-%m-%d)} +GPU=${GPU:-0} +MODEL_ID=openbmb/MiniCPM-V-4_5 + +cd "$REPO_DIR" +mkdir -p "$OUT_ROOT" + +for spec in "no_media|32|" "single_image_start|64|tests/media/448"; do + IFS='|' read -r sample maxtok imgdir <<< "$spec" + out="$OUT_ROOT/$sample" + extra=() + [[ -n "$imgdir" ]] && extra=(--img_dir "$imgdir") + PYTHONPATH="$REPO_DIR" ENV_DIR=/raid/ztw/envs CUDA_VISIBLE_DEVICES="$GPU" \ + /raid/ztw/envs/minicpm_v_4d5/bin/python mmeval/run.py \ + --model_name_or_path "$MODEL_ID" \ + --dataset local@json --infile tests/samples/$sample.json \ + --out_dir "$out" \ + --gpu_per_parallel 1 --parallel_per_task 1 --max_new_tokens "$maxtok" \ + "${extra[@]}" +done