diff --git a/env_files/glm_4d1v_requirements.txt b/env_files/glm_4d1v_requirements.txt new file mode 100644 index 00000000..6004f8af --- /dev/null +++ b/env_files/glm_4d1v_requirements.txt @@ -0,0 +1,12 @@ +torch==2.6.0 +torchvision==0.21.0 +transformers==4.57.1 +accelerate +datasets +pandas +numpy==1.26.4 +Pillow +sentencepiece +protobuf +einops +timm diff --git a/mmeval/infer/glm_4d1v.py b/mmeval/infer/glm_4d1v.py new file mode 100644 index 00000000..6f666adf --- /dev/null +++ b/mmeval/infer/glm_4d1v.py @@ -0,0 +1,96 @@ +"""glm_4d1v — GLM-4.1V-9B-Thinking. + +HF: https://huggingface.co/THUDM/GLM-4.1V-9B-Thinking +GH: https://github.com/THUDM/GLM-V +""" +import re +import copy + +import torch +from PIL import Image +from transformers import AutoProcessor, Glm4vForConditionalGeneration + +from mmeval.infer.task import Task +from mmeval.utils import constants +from mmeval.utils.argparser import parse_args, parse_model_kwargs, parse_gen_kwargs + + +class TaskRunner(Task): + def __init__(self, args): + self.args = args + self.device = torch.device("cuda" if torch.cuda.is_available() else "cpu") + self.dtype = getattr(args, "dtype") or torch.bfloat16 + self.default_model_kwargs = {"device_map": "auto"} + self.default_gen_kwargs = {"max_new_tokens": 8192} + self.model_kwargs = parse_model_kwargs(args, self.default_model_kwargs) + self.gen_kwargs = parse_gen_kwargs(args, self.default_gen_kwargs) + + super().__init__(args) + + def load_model(self, args): + self.model = Glm4vForConditionalGeneration.from_pretrained( + args.model_name_or_path, + torch_dtype=self.dtype, + **self.model_kwargs, + ).eval() + self.processor = AutoProcessor.from_pretrained( + args.model_name_or_path, use_fast=True, + ) + + def parse_input(self, message): + question = message["prompt"] + q_chunks = re.split(r'(<(?:image|video)>)', question) + media_list = message.get('media', []) + + content = [] + media_idx = 0 + for chunk in q_chunks: + if not chunk.strip(): + continue + if chunk == constants.image: + img = media_list[media_idx] + if isinstance(img, str): + img = Image.open(img).convert("RGB") + elif hasattr(img, "convert"): + img = img.convert("RGB") + content.append({"type": "image", "image": img}) + media_idx += 1 + elif chunk == constants.video: + content.append({"type": "video", "video": media_list[media_idx]}) + media_idx += 1 + else: + content.append({"type": "text", "text": chunk}) + return [{"role": "user", "content": content}] + + def _generate_response(self, inputs): + generated_ids = self.model.generate(**inputs, **self.gen_kwargs) + trimmed = generated_ids[0][inputs["input_ids"].shape[1]:] + return self.processor.decode(trimmed, skip_special_tokens=False) + + def run_sample(self, sample: dict): + if self.args.score_target: + raise NotImplementedError( + "glm_4d1v: score_target is not implemented yet" + ) + + ori_sample = copy.deepcopy(sample) + message = sample["messages"][0] + messages = self.parse_input(message) + + inputs = self.processor.apply_chat_template( + messages, + tokenize=True, + add_generation_prompt=True, + return_dict=True, + return_tensors="pt", + ).to(self.model.device) + + response = self._generate_response(inputs) + ori_sample["messages"].append({"role": "assistant", "response": response}) + return ori_sample + + +if __name__ == "__main__": + args = parse_args() + model_evaluator = TaskRunner(args) + model_evaluator.inference_dataset() diff --git a/mmeval/registry.py b/mmeval/registry.py index dfbf0b98..cc3160d2 100644 --- a/mmeval/registry.py +++ b/mmeval/registry.py @@ -77,6 +77,7 @@ "doubao-seed-2-0-mini-260215", "doubao-seed-2-0-lite-260215", "doubao-seed-2-0-code-preview-260215", "doubao-seed-2-0-pro-260215"], "hunyuan_vision": ["hunyuan-vision", "hunyuan-vision-1.5-instruct", "hunyuan-t1-vision", "hunyuan-turbos-vision", "hunyuan-large-vision"], "cosmos_reason2": ["Cosmos-Reason2-2B", "Cosmos-Reason2-8B"], + "glm_4d1v": ["GLM-4.1V-9B-Thinking", "GLM-4.1V-9B-Base"], } series_infer_env_mapping = { @@ -316,4 +317,8 @@ "env": os.path.join(env_dir, "cosmos_reason2"), "infer_file": "cosmos_reason2.py", }, + "glm_4d1v": { + "env": os.path.join(env_dir, "glm_4d1v"), + "infer_file": "glm_4d1v.py", + }, } diff --git a/test_results/glm_4d1v_2026-06-10/no_media/result.json b/test_results/glm_4d1v_2026-06-10/no_media/result.json new file mode 100644 index 00000000..040588f2 --- /dev/null +++ b/test_results/glm_4d1v_2026-06-10/no_media/result.json @@ -0,0 +1,44 @@ +[ + { + "id": 0, + "media": [], + "messages": [ + { + "role": "user", + "question": "What is the result of 1 plus 1?", + "answer": "", + "options": {}, + "choices": [], + "prompt": "What is the result of 1 plus 1?", + "hint": "" + }, + { + "role": "assistant", + "response": "Got it, let's see. The question is asking for the result of 1 plus 1. That's a basic arithmetic operation. When you add" + } + ], + "comment": "Test 'no media' modality.", + "eval-id": 0 + }, + { + "id": 1, + "media": [], + "messages": [ + { + "role": "user", + "question": "What is the capital city of China?", + "answer": "", + "options": {}, + "choices": [], + "prompt": "What is the capital city of China?", + "hint": "" + }, + { + "role": "assistant", + "response": "Got it, the question is asking for the capital city of China. I need to recall that the capital of China is Beijing. Let me make sure there" + } + ], + "comment": "Test 'no media' modality.", + "eval-id": 1 + } +] \ No newline at end of file diff --git a/test_results/glm_4d1v_2026-06-10/single_image_start/result.json b/test_results/glm_4d1v_2026-06-10/single_image_start/result.json new file mode 100644 index 00000000..544174c4 --- /dev/null +++ b/test_results/glm_4d1v_2026-06-10/single_image_start/result.json @@ -0,0 +1,48 @@ +[ + { + "id": 0, + "media": [ + "truck.png" + ], + "messages": [ + { + "role": "user", + "question": " Please provide a detailed description of the contents shown in the image.", + "answer": "", + "options": {}, + "choices": [], + "prompt": " Please provide a detailed description of the contents shown in the image.", + "hint": "" + }, + { + "role": "assistant", + "response": "Got it, let's start by describing the image. The main subject is a GMC pickup truck, specifically a modern model, probably the 2024 GMC Hummer EV or a similar model, but wait, no, the design looks like the GMC Sierra or maybe the Hummer pickup? Wait, the image shows" + } + ], + "comment": "Test 'single image start' modality.", + "eval-id": 0 + }, + { + "id": 1, + "media": [ + "boat.png" + ], + "messages": [ + { + "role": "user", + "question": " Please provide a detailed description of the contents shown in the image.", + "answer": "", + "options": {}, + "choices": [], + "prompt": " Please provide a detailed description of the contents shown in the image.", + "hint": "" + }, + { + "role": "assistant", + "response": "Got it, let's analyze the image. The main subject is a large white yacht or motorboat moving through the water. The boat is sleek, with a modern design, probably a luxury yacht. It's creating a big wake as it moves, which shows it's going at a good speed. The water is" + } + ], + "comment": "Test 'single image start' modality.", + "eval-id": 1 + } +] \ No newline at end of file diff --git a/test_results/glm_4d1v_2026-06-10/test_summary.json b/test_results/glm_4d1v_2026-06-10/test_summary.json new file mode 100644 index 00000000..f9e85804 --- /dev/null +++ b/test_results/glm_4d1v_2026-06-10/test_summary.json @@ -0,0 +1,20 @@ +{ + "model_series": "glm_4d1v", + "model_name": "THUDM/GLM-4.1V-9B-Thinking", + "test_date": "2026-06-10", + "conda_env": "/raid/ztw/envs/glm_4d1v", + "env_summary": "transformers==4.57.1", + "status": "PASS", + "modalities_tested": { + "no_media": { + "status": "pass", + "rows": 2, + "result_path": "/raid/ztw/simple-mmeval-test-result/work_dirs/glm_4d1v/2026-06-10/no_media/result.json" + }, + "single_image_start": { + "status": "pass", + "rows": 2, + "result_path": "/raid/ztw/simple-mmeval-test-result/work_dirs/glm_4d1v/2026-06-10/single_image_start/result.json" + } + } +} \ No newline at end of file diff --git a/test_results/test_glm_4d1v.sh b/test_results/test_glm_4d1v.sh new file mode 100755 index 00000000..1cb90b67 --- /dev/null +++ b/test_results/test_glm_4d1v.sh @@ -0,0 +1,25 @@ +#!/usr/bin/env bash +# Smoke test for glm_4d1v (model: THUDM/GLM-4.1V-9B-Thinking). +set -euo pipefail + +REPO_DIR=${REPO_DIR:-/raid/ztw/simple-mmeval-skills-dev} +OUT_ROOT=${OUT_ROOT:-/raid/ztw/simple-mmeval-test-result/work_dirs/glm_4d1v/$(date +%Y-%m-%d)} +GPU=${GPU:-0} +MODEL_ID=THUDM/GLM-4.1V-9B-Thinking + +cd "$REPO_DIR" +mkdir -p "$OUT_ROOT" + +for spec in "no_media|32|" "single_image_start|64|tests/media/448"; do + IFS='|' read -r sample maxtok imgdir <<< "$spec" + out="$OUT_ROOT/$sample" + extra=() + [[ -n "$imgdir" ]] && extra=(--img_dir "$imgdir") + PYTHONPATH="$REPO_DIR" ENV_DIR=/raid/ztw/envs CUDA_VISIBLE_DEVICES="$GPU" \ + /raid/ztw/envs/glm_4d1v/bin/python mmeval/run.py \ + --model_name_or_path "$MODEL_ID" \ + --dataset local@json --infile tests/samples/$sample.json \ + --out_dir "$out" \ + --gpu_per_parallel 1 --parallel_per_task 1 --max_new_tokens "$maxtok" \ + "${extra[@]}" +done