diff --git a/PyTorch/built-in/mlm/MiniCPM-V/.gitignore b/PyTorch/built-in/mlm/MiniCPM-V/.gitignore
new file mode 100644
index 0000000000000000000000000000000000000000..988f840d44e6a358f3fe196ec8006bcef49217dc
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/.gitignore
@@ -0,0 +1,3 @@
+*.bk
+__pycache__
+.DS_Store
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/LICENSE b/PyTorch/built-in/mlm/MiniCPM-V/LICENSE
new file mode 100644
index 0000000000000000000000000000000000000000..2756a35237431c301015ac7c326cc0120dd4965d
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/LICENSE
@@ -0,0 +1,201 @@
+                                 Apache License
+                           Version 2.0, January 2004
+                        http://www.apache.org/licenses/
+
+   TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
+
+   1. Definitions.
+
+      "License" shall mean the terms and conditions for use, reproduction,
+      and distribution as defined by Sections 1 through 9 of this document.
+
+      "Licensor" shall mean the copyright owner or entity authorized by
+      the copyright owner that is granting the License.
+
+      "Legal Entity" shall mean the union of the acting entity and all
+      other entities that control, are controlled by, or are under common
+      control with that entity. For the purposes of this definition,
+      "control" means (i) the power, direct or indirect, to cause the
+      direction or management of such entity, whether by contract or
+      otherwise, or (ii) ownership of fifty percent (50%) or more of the
+      outstanding shares, or (iii) beneficial ownership of such entity.
+
+      "You" (or "Your") shall mean an individual or Legal Entity
+      exercising permissions granted by this License.
+
+      "Source" form shall mean the preferred form for making modifications,
+      including but not limited to software source code, documentation
+      source, and configuration files.
+
+      "Object" form shall mean any form resulting from mechanical
+      transformation or translation of a Source form, including but
+      not limited to compiled object code, generated documentation,
+      and conversions to other media types.
+
+      "Work" shall mean the work of authorship, whether in Source or
+      Object form, made available under the License, as indicated by a
+      copyright notice that is included in or attached to the work
+      (an example is provided in the Appendix below).
+
+      "Derivative Works" shall mean any work, whether in Source or Object
+      form, that is based on (or derived from) the Work and for which the
+      editorial revisions, annotations, elaborations, or other modifications
+      represent, as a whole, an original work of authorship. For the purposes
+      of this License, Derivative Works shall not include works that remain
+      separable from, or merely link (or bind by name) to the interfaces of,
+      the Work and Derivative Works thereof.
+
+      "Contribution" shall mean any work of authorship, including
+      the original version of the Work and any modifications or additions
+      to that Work or Derivative Works thereof, that is intentionally
+      submitted to Licensor for inclusion in the Work by the copyright owner
+      or by an individual or Legal Entity authorized to submit on behalf of
+      the copyright owner. For the purposes of this definition, "submitted"
+      means any form of electronic, verbal, or written communication sent
+      to the Licensor or its representatives, including but not limited to
+      communication on electronic mailing lists, source code control systems,
+      and issue tracking systems that are managed by, or on behalf of, the
+      Licensor for the purpose of discussing and improving the Work, but
+      excluding communication that is conspicuously marked or otherwise
+      designated in writing by the copyright owner as "Not a Contribution."
+
+      "Contributor" shall mean Licensor and any individual or Legal Entity
+      on behalf of whom a Contribution has been received by Licensor and
+      subsequently incorporated within the Work.
+
+   2. Grant of Copyright License. Subject to the terms and conditions of
+      this License, each Contributor hereby grants to You a perpetual,
+      worldwide, non-exclusive, no-charge, royalty-free, irrevocable
+      copyright license to reproduce, prepare Derivative Works of,
+      publicly display, publicly perform, sublicense, and distribute the
+      Work and such Derivative Works in Source or Object form.
+
+   3. Grant of Patent License. Subject to the terms and conditions of
+      this License, each Contributor hereby grants to You a perpetual,
+      worldwide, non-exclusive, no-charge, royalty-free, irrevocable
+      (except as stated in this section) patent license to make, have made,
+      use, offer to sell, sell, import, and otherwise transfer the Work,
+      where such license applies only to those patent claims licensable
+      by such Contributor that are necessarily infringed by their
+      Contribution(s) alone or by combination of their Contribution(s)
+      with the Work to which such Contribution(s) was submitted. If You
+      institute patent litigation against any entity (including a
+      cross-claim or counterclaim in a lawsuit) alleging that the Work
+      or a Contribution incorporated within the Work constitutes direct
+      or contributory patent infringement, then any patent licenses
+      granted to You under this License for that Work shall terminate
+      as of the date such litigation is filed.
+
+   4. Redistribution. You may reproduce and distribute copies of the
+      Work or Derivative Works thereof in any medium, with or without
+      modifications, and in Source or Object form, provided that You
+      meet the following conditions:
+
+      (a) You must give any other recipients of the Work or
+          Derivative Works a copy of this License; and
+
+      (b) You must cause any modified files to carry prominent notices
+          stating that You changed the files; and
+
+      (c) You must retain, in the Source form of any Derivative Works
+          that You distribute, all copyright, patent, trademark, and
+          attribution notices from the Source form of the Work,
+          excluding those notices that do not pertain to any part of
+          the Derivative Works; and
+
+      (d) If the Work includes a "NOTICE" text file as part of its
+          distribution, then any Derivative Works that You distribute must
+          include a readable copy of the attribution notices contained
+          within such NOTICE file, excluding those notices that do not
+          pertain to any part of the Derivative Works, in at least one
+          of the following places: within a NOTICE text file distributed
+          as part of the Derivative Works; within the Source form or
+          documentation, if provided along with the Derivative Works; or,
+          within a display generated by the Derivative Works, if and
+          wherever such third-party notices normally appear. The contents
+          of the NOTICE file are for informational purposes only and
+          do not modify the License. You may add Your own attribution
+          notices within Derivative Works that You distribute, alongside
+          or as an addendum to the NOTICE text from the Work, provided
+          that such additional attribution notices cannot be construed
+          as modifying the License.
+
+      You may add Your own copyright statement to Your modifications and
+      may provide additional or different license terms and conditions
+      for use, reproduction, or distribution of Your modifications, or
+      for any such Derivative Works as a whole, provided Your use,
+      reproduction, and distribution of the Work otherwise complies with
+      the conditions stated in this License.
+
+   5. Submission of Contributions. Unless You explicitly state otherwise,
+      any Contribution intentionally submitted for inclusion in the Work
+      by You to the Licensor shall be under the terms and conditions of
+      this License, without any additional terms or conditions.
+      Notwithstanding the above, nothing herein shall supersede or modify
+      the terms of any separate license agreement you may have executed
+      with Licensor regarding such Contributions.
+
+   6. Trademarks. This License does not grant permission to use the trade
+      names, trademarks, service marks, or product names of the Licensor,
+      except as required for reasonable and customary use in describing the
+      origin of the Work and reproducing the content of the NOTICE file.
+
+   7. Disclaimer of Warranty. Unless required by applicable law or
+      agreed to in writing, Licensor provides the Work (and each
+      Contributor provides its Contributions) on an "AS IS" BASIS,
+      WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
+      implied, including, without limitation, any warranties or conditions
+      of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
+      PARTICULAR PURPOSE. You are solely responsible for determining the
+      appropriateness of using or redistributing the Work and assume any
+      risks associated with Your exercise of permissions under this License.
+
+   8. Limitation of Liability. In no event and under no legal theory,
+      whether in tort (including negligence), contract, or otherwise,
+      unless required by applicable law (such as deliberate and grossly
+      negligent acts) or agreed to in writing, shall any Contributor be
+      liable to You for damages, including any direct, indirect, special,
+      incidental, or consequential damages of any character arising as a
+      result of this License or out of the use or inability to use the
+      Work (including but not limited to damages for loss of goodwill,
+      work stoppage, computer failure or malfunction, or any and all
+      other commercial damages or losses), even if such Contributor
+      has been advised of the possibility of such damages.
+
+   9. Accepting Warranty or Additional Liability. While redistributing
+      the Work or Derivative Works thereof, You may choose to offer,
+      and charge a fee for, acceptance of support, warranty, indemnity,
+      or other liability obligations and/or rights consistent with this
+      License. However, in accepting such obligations, You may act only
+      on Your own behalf and on Your sole responsibility, not on behalf
+      of any other Contributor, and only if You agree to indemnify,
+      defend, and hold each Contributor harmless for any liability
+      incurred by, or claims asserted against, such Contributor by reason
+      of your accepting any such warranty or additional liability.
+
+   END OF TERMS AND CONDITIONS
+
+   APPENDIX: How to apply the Apache License to your work.
+
+      To apply the Apache License to your work, attach the following
+      boilerplate notice, with the fields enclosed by brackets "[]"
+      replaced with your own identifying information. (Don't include
+      the brackets!)  The text should be enclosed in the appropriate
+      comment syntax for the file format. We also recommend that a
+      file or class name and description of purpose be included on the
+      same "printed page" as the copyright notice for easier
+      identification within third-party archives.
+
+   Copyright 2024 OpenBMB
+
+   Licensed under the Apache License, Version 2.0 (the "License");
+   you may not use this file except in compliance with the License.
+   You may obtain a copy of the License at
+
+       http://www.apache.org/licenses/LICENSE-2.0
+
+   Unless required by applicable law or agreed to in writing, software
+   distributed under the License is distributed on an "AS IS" BASIS,
+   WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+   See the License for the specific language governing permissions and
+   limitations under the License.
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/chat.py b/PyTorch/built-in/mlm/MiniCPM-V/chat.py
new file mode 100644
index 0000000000000000000000000000000000000000..8dbf8ef01a98a2afd6e77407653aec41fbb70fb9
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/chat.py
@@ -0,0 +1,219 @@
+import os
+import torch
+import json
+from PIL import Image
+import base64
+import io
+from accelerate import load_checkpoint_and_dispatch, init_empty_weights
+from transformers import AutoTokenizer, AutoModel
+
+from omnilmm.utils import disable_torch_init
+from omnilmm.model.omnilmm import OmniLMMForCausalLM
+from omnilmm.model.utils import build_transform
+from omnilmm.train.train_utils import omni_preprocess
+
+DEFAULT_IMAGE_TOKEN = "<image>"
+DEFAULT_IMAGE_PATCH_TOKEN = "<im_patch>"
+DEFAULT_IM_START_TOKEN = "<im_start>"
+DEFAULT_IM_END_TOKEN = "<im_end>"
+
+    
+
+def init_omni_lmm(model_path):
+    torch.backends.cuda.matmul.allow_tf32 = True
+    disable_torch_init()
+    model_name = os.path.expanduser(model_path)
+    print(f'Load omni_lmm model and tokenizer from {model_name}')
+    tokenizer = AutoTokenizer.from_pretrained(
+        model_name, model_max_length=2048)
+
+    if False:
+        # model on multiple devices for small size gpu memory (Nvidia 3090 24G x2) 
+        with init_empty_weights():
+            model = OmniLMMForCausalLM.from_pretrained(model_name, tune_clip=True, torch_dtype=torch.bfloat16)
+        model = load_checkpoint_and_dispatch(model, model_name, dtype=torch.bfloat16, 
+                    device_map="auto",  no_split_module_classes=['Eva','MistralDecoderLayer', 'ModuleList', 'Resampler']
+        )
+    else:
+        model = OmniLMMForCausalLM.from_pretrained(
+            model_name, tune_clip=True, torch_dtype=torch.bfloat16
+        ).to(device='cuda', dtype=torch.bfloat16)
+
+    image_processor = build_transform(
+        is_train=False, input_size=model.model.config.image_size, std_mode='OPENAI_CLIP')
+
+    mm_use_im_start_end = getattr(model.config, "mm_use_im_start_end", False)
+    assert mm_use_im_start_end
+
+    tokenizer.add_tokens([DEFAULT_IMAGE_PATCH_TOKEN, DEFAULT_IM_START_TOKEN,
+                         DEFAULT_IM_END_TOKEN], special_tokens=True)
+
+
+    vision_config = model.model.vision_config
+    vision_config.im_patch_token = tokenizer.convert_tokens_to_ids(
+        [DEFAULT_IMAGE_PATCH_TOKEN])[0]
+    vision_config.use_im_start_end = mm_use_im_start_end
+    vision_config.im_start_token, vision_config.im_end_token = tokenizer.convert_tokens_to_ids(
+        [DEFAULT_IM_START_TOKEN, DEFAULT_IM_END_TOKEN])
+    image_token_len = model.model.config.num_query
+
+    return model, image_processor, image_token_len, tokenizer
+
+def expand_question_into_multimodal(question_text, image_token_len, im_st_token, im_ed_token, im_patch_token):
+    if '<image>' in question_text[0]['content']:
+        question_text[0]['content'] = question_text[0]['content'].replace(
+            '<image>', im_st_token + im_patch_token * image_token_len + im_ed_token)
+    else:
+        question_text[0]['content'] = im_st_token + im_patch_token * \
+            image_token_len + im_ed_token + '\n' + question_text[0]['content']
+    return question_text
+
+def wrap_question_for_omni_lmm(question, image_token_len, tokenizer):
+    question = expand_question_into_multimodal(
+        question, image_token_len, DEFAULT_IM_START_TOKEN, DEFAULT_IM_END_TOKEN, DEFAULT_IMAGE_PATCH_TOKEN)
+
+    conversation = question
+    data_dict = omni_preprocess(sources=[conversation],
+                                  tokenizer=tokenizer,
+                                  generation=True)
+
+    data_dict = dict(input_ids=data_dict["input_ids"][0],
+                     labels=data_dict["labels"][0])
+    return data_dict
+
+
+
+class OmniLMM12B:
+    def __init__(self, model_path) -> None:
+        model, img_processor, image_token_len, tokenizer = init_omni_lmm(model_path)
+        self.model = model
+        self.image_token_len = image_token_len
+        self.image_transform = img_processor
+        self.tokenizer = tokenizer
+        self.model.eval()
+
+    def decode(self, image, input_ids):
+        with torch.inference_mode():
+            output = self.model.generate_vllm(
+                input_ids=input_ids.unsqueeze(0).cuda(),
+                images=image.unsqueeze(0).half().cuda(),
+                temperature=0.6,
+                max_new_tokens=1024,
+                # num_beams=num_beams,
+                do_sample=True,
+                output_scores=True,
+                return_dict_in_generate=True,
+                repetition_penalty=1.1,
+                top_k=30,
+                top_p=0.9,
+            )
+
+            response = self.tokenizer.decode(
+                output.sequences[0], skip_special_tokens=True)
+            response = response.strip()
+            return response
+
+    def chat(self, input):
+        try:
+            image = Image.open(io.BytesIO(base64.b64decode(input['image']))).convert('RGB')
+        except Exception as e:
+            return "Image decode error"
+
+        msgs = json.loads(input['question'])
+        input_ids = wrap_question_for_omni_lmm(
+            msgs, self.image_token_len, self.tokenizer)['input_ids']
+        input_ids = torch.as_tensor(input_ids)
+        #print('input_ids', input_ids)
+        image = self.image_transform(image)
+
+        out = self.decode(image, input_ids)
+
+        return out
+        
+
+def img2base64(file_name):
+    with open(file_name, 'rb') as f:
+        encoded_string = base64.b64encode(f.read())
+        return encoded_string
+
+class MiniCPMV:
+    def __init__(self, model_path) -> None:
+        self.model = AutoModel.from_pretrained(model_path, trust_remote_code=True).to(dtype=torch.bfloat16)
+        self.tokenizer = AutoTokenizer.from_pretrained(model_path, trust_remote_code=True)
+        self.model.eval().cuda()
+
+    def chat(self, input):
+        try:
+            image = Image.open(io.BytesIO(base64.b64decode(input['image']))).convert('RGB')
+        except Exception as e:
+            return "Image decode error"
+
+        msgs = json.loads(input['question'])
+        
+        answer, context, _ = self.model.chat(
+            image=image,
+            msgs=msgs,
+            context=None,
+            tokenizer=self.tokenizer,
+            sampling=True,
+            temperature=0.7
+    	)
+        return answer
+
+class MiniCPMV2_5:
+    def __init__(self, model_path) -> None:
+        self.model = AutoModel.from_pretrained(model_path, trust_remote_code=True).to(dtype=torch.float16)
+        self.tokenizer = AutoTokenizer.from_pretrained(model_path, trust_remote_code=True)
+        self.model.eval().cuda()
+
+    def chat(self, input):
+        try:
+            image = Image.open(io.BytesIO(base64.b64decode(input['image']))).convert('RGB')
+        except Exception as e:
+            return "Image decode error"
+
+        msgs = json.loads(input['question'])
+        
+        answer = self.model.chat(
+            image=image,
+            msgs=msgs,
+            tokenizer=self.tokenizer,
+            sampling=True,
+            temperature=0.7
+    	)
+        return answer
+
+
+class MiniCPMVChat:
+    def __init__(self, model_path) -> None:
+        if '12B' in model_path:
+            self.model = OmniLMM12B(model_path)
+        elif 'MiniCPM-Llama3-V' in model_path:
+            self.model = MiniCPMV2_5(model_path)
+        else:
+            self.model = MiniCPMV(model_path)
+
+    def chat(self, input):
+        return self.model.chat(input)
+
+
+if __name__ == '__main__':
+    
+    model_path = 'openbmb/OmniLMM-12B'
+    chat_model = MiniCPMVChat(model_path)
+
+    im_64 = img2base64('./assets/worldmap_ck.jpg')
+
+    # first round chat 
+    msgs = [{"role": "user", "content": "What is interesting about this image?"}]
+    input = {"image": im_64, "question": json.dumps(msgs, ensure_ascii=True)}
+    answer = chat_model.chat(input)
+    print(msgs[-1]["content"]+'\n', answer)
+
+    # second round chat 
+    msgs.append({"role": "assistant", "content": answer})
+    msgs.append({"role": "user", "content": "Where is China in the image"})
+    input = {"image": im_64,"question": json.dumps(msgs, ensure_ascii=True)}
+    answer = chat_model.chat(input)
+    print(msgs[-1]["content"]+'\n', answer)
+
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/README.md b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..c094716dea2aa179437d6a2c9508c21c40472f85
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/README.md
@@ -0,0 +1,177 @@
+# Evaluation
+
+## opencompass
+First, enter the `vlmevalkit` directory and install all dependencies:
+```bash
+cd vlmevalkit
+pip install -r requirements.txt
+```
+<br />
+
+Then, run `script/run_inference.sh`, which receives three input parameters in sequence: `MODELNAME`, `DATALIST`, and `MODE`. `MODELNAME` represents the name of the model, `DATALIST` represents the datasets used for inference, and `MODE` represents evaluation mode:
+```bash
+chmod +x ./script/run_inference.sh
+./script/run_inference.sh $MODELNAME $DATALIST $MODE
+```
+<br />
+
+The three available choices for `MODELNAME` are listed in `vlmeval/config.py`:
+```bash
+ungrouped = {
+    'MiniCPM-V':partial(MiniCPM_V, model_path='openbmb/MiniCPM-V'),
+    'MiniCPM-V-2':partial(MiniCPM_V, model_path='openbmb/MiniCPM-V-2'),
+    'MiniCPM-Llama3-V-2_5':partial(MiniCPM_Llama3_V, model_path='openbmb/MiniCPM-Llama3-V-2_5'),
+}
+```
+<br />
+
+All available choices for `DATALIST` are listed in `vlmeval/utils/dataset_config.py`. While evaluating on a single dataset, call the dataset name directly without quotation marks; while evaluating on multiple datasets, separate the names of different datasets with spaces and add quotation marks at both ends:
+```bash
+$DATALIST="POPE ScienceQA_TEST ChartQA_TEST"
+```
+<br />
+
+While scoring on each benchmark directly, set `MODE=all`. If only inference results are required, set `MODE=infer`. In order to reproduce the results in the table displayed on the homepage (columns between MME and RealWorldQA), you need to run the script according to the following settings:
+```bash
+# run on all 7 datasets
+./script/run_inference.sh MiniCPM-Llama3-V-2_5 "MME MMBench_TEST_EN MMBench_TEST_CN MMMU_DEV_VAL MathVista_MINI LLaVABench RealWorldQA" all
+
+# The following are instructions for running on a single dataset
+# MME
+./script/run_inference.sh MiniCPM-Llama3-V-2_5 MME all
+# MMBench_TEST_EN
+./script/run_inference.sh MiniCPM-Llama3-V-2_5 MMBench_TEST_EN all
+# MMBench_TEST_CN
+./script/run_inference.sh MiniCPM-Llama3-V-2_5 MMBench_TEST_CN all
+# MMMU_DEV_VAL
+./script/run_inference.sh MiniCPM-Llama3-V-2_5 MMMU_DEV_VAL all
+# MathVista_MINI
+./script/run_inference.sh MiniCPM-Llama3-V-2_5 MathVista_MINI all
+# LLaVABench
+./script/run_inference.sh MiniCPM-Llama3-V-2_5 LLaVABench all
+# RealWorldQA
+./script/run_inference.sh MiniCPM-Llama3-V-2_5 RealWorldQA all
+```
+<br />
+
+## vqadataset
+First, enter the `vqaeval` directory and install all dependencies. Then, create `downloads` subdirectory to store the downloaded dataset for all tasks:
+```bash
+cd vqaeval
+pip install -r requirements.txt
+mkdir downloads
+```
+<br />
+
+Download the datasets from the following links and place it in the specified directories:
+###### TextVQA
+```bash
+cd downloads
+mkdir TextVQA && cd TextVQA
+wget https://dl.fbaipublicfiles.com/textvqa/images/train_val_images.zip
+unzip train_val_images.zip && rm train_val_images.zip
+mv train_val_images/train_images . && rm -rf train_val_images
+wget https://dl.fbaipublicfiles.com/textvqa/data/TextVQA_0.5.1_val.json
+cd ../..
+```
+
+###### DocVQA / DocVQATest
+
+```bash
+cd downloads
+mkdir DocVQA && cd DocVQA && mkdir spdocvqa_images
+# Download Images and Annotations from Task 1 - Single Page Document Visual Question Answering at https://rrc.cvc.uab.es/?ch=17&com=downloads
+# Move the spdocvqa_images.tar.gz and spdocvqa_qas.zip to DocVQA directory
+tar -zxvf spdocvqa_images.tar.gz -C spdocvqa_images && rm spdocvqa_images.tar.gz
+unzip spdocvqa_qas.zip && rm spdocvqa_qas.zip
+cp spdocvqa_qas/val_v1.0_withQT.json . && cp spdocvqa_qas/test_v1.0.json .  && rm -rf spdocvqa_qas
+cd ../..
+```
+<br />
+
+The `downloads` directory should be organized according to the following structure:
+```bash
+downloads
+├── TextVQA
+│   ├── train_images
+│   │   ├── ...
+│   ├── TextVQA_0.5.1_val.json
+├── DocVQA
+│   ├── spdocvqa_images
+│   │   ├── ...
+│   ├── val_v1.0_withQT.json
+│   ├── test_v1.0.json
+```
+<br />
+
+Modify the parameters in `shell/run_inference.sh` and run inference:
+
+```bash
+chmod +x ./shell/run_inference.sh
+./shell/run_inference.sh
+```
+<br />
+
+All optional parameters are listed in `eval_utils/getargs.py`. The meanings of some major parameters are listed as follows:
+```bash
+# path to images and their corresponding questions
+# TextVQA
+--textVQA_image_dir
+--textVQA_ann_path
+# DocVQA
+--docVQA_image_dir
+--docVQA_ann_path
+# DocVQATest
+--docVQATest_image_dir
+--docVQATest_ann_path
+
+# whether to eval on certain task
+--eval_textVQA
+--eval_docVQA
+--eval_docVQATest
+--eval_all
+
+# model name and model path
+--model_name
+--model_path
+# load model from ckpt
+--ckpt
+# the way the model processes input data, "interleave" represents interleaved image-text form, while "old" represents non-interleaved.
+--generate_method
+
+--batchsize
+
+# path to save the outputs
+--answer_path
+```
+<br />
+
+While evaluating on different tasks, parameters need to be set as follows:
+###### TextVQA
+```bash
+--eval_textVQA
+--textVQA_image_dir ./downloads/TextVQA/train_images
+--textVQA_ann_path ./downloads/TextVQA/TextVQA_0.5.1_val.json
+```
+
+###### DocVQA
+```bash
+--eval_docVQA
+--docVQA_image_dir ./downloads/DocVQA/spdocvqa_images
+--docVQA_ann_path ./downloads/DocVQA/val_v1.0_withQT.json
+```
+
+###### DocVQATest
+```bash
+--eval_docVQATest
+--docVQATest_image_dir ./downloads/DocVQA/spdocvqa_images
+--docVQATest_ann_path ./downloads/DocVQA/test_v1.0.json
+```
+
+<br />
+
+For the DocVQATest task, in order to upload the inference results to the [official website](https://rrc.cvc.uab.es/?ch=17) for evaluation, run `shell/run_transform.sh` for format transformation after inference. `input_file_path` represents the path to the original output json, `output_file_path` represents the path to the transformed json:
+```bash
+chmod +x ./shell/run_transform.sh
+./shell/run_transform.sh
+```
\ No newline at end of file
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/README_zh.md b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/README_zh.md
new file mode 100644
index 0000000000000000000000000000000000000000..58e0288a3a348f6ec34a8363b81d89106a98f620
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/README_zh.md
@@ -0,0 +1,175 @@
+# Evaluation
+
+## opencompass
+首先，进入 `vlmevalkit` 目录下，安装必要的依赖：
+```bash
+cd vlmevalkit
+pip install -r requirements.txt
+```
+<br />
+
+然后，运行 `script/run_inference.sh`，该脚本依次接收三个输入参数：`MODELNAME`, `DATALIST`, `MODE`。`MODELNAME` 为模型名称，`DATALIST` 为目标数据集，`MODE` 为评测模式。
+```bash
+chmod +x ./script/run_inference.sh
+./script/run_inference.sh $MODELNAME $DATALIST $MODE
+```
+<br />
+
+`MODELNAME` 有三种选择，位于 `vlmeval/config.py` 中：
+```bash
+ungrouped = {
+    'MiniCPM-V':partial(MiniCPM_V, model_path='openbmb/MiniCPM-V'),
+    'MiniCPM-V-2':partial(MiniCPM_V, model_path='openbmb/MiniCPM-V-2'),
+    'MiniCPM-Llama3-V-2_5':partial(MiniCPM_Llama3_V, model_path='openbmb/MiniCPM-Llama3-V-2_5'),
+}
+```
+<br />
+
+可选的所有 `DATALIST` 位于 `vlmeval/utils/dataset_config.py` 中，评测单个数据集时，直接调用数据集名称，不加引号；评测多个数据集时，将不同数据集名称以空格隔开，两端加引号：
+```bash
+$DATALIST="POPE ScienceQA_TEST ChartQA_TEST"
+```
+<br />
+
+直接对各 benchmark 进行评分时，设置 `MODE=all`。如果仅需要推理结果，则设置 `MODE=infer`
+为了复现出首页展示的表格中的各项结果（MME 到 RealWorldQA 之间的列），需要按照如下设置运行：
+```bash
+# 一次性运行 7 个数据集
+./script/run_inference.sh MiniCPM-Llama3-V-2_5 "MME MMBench_TEST_EN MMBench_TEST_CN MMMU_DEV_VAL MathVista_MINI LLaVABench RealWorldQA" all
+
+# 以下是单独运行 1 个数据集的指令
+# MME
+./script/run_inference.sh MiniCPM-Llama3-V-2_5 MME all
+# MMBench_TEST_EN
+./script/run_inference.sh MiniCPM-Llama3-V-2_5 MMBench_TEST_EN all
+# MMBench_TEST_CN
+./script/run_inference.sh MiniCPM-Llama3-V-2_5 MMBench_TEST_CN all
+# MMMU_DEV_VAL
+./script/run_inference.sh MiniCPM-Llama3-V-2_5 MMMU_DEV_VAL all
+# MathVista_MINI
+./script/run_inference.sh MiniCPM-Llama3-V-2_5 MathVista_MINI all
+# LLaVABench
+./script/run_inference.sh MiniCPM-Llama3-V-2_5 LLaVABench all
+# RealWorldQA
+./script/run_inference.sh MiniCPM-Llama3-V-2_5 RealWorldQA all
+```
+<br />
+
+## vqadataset
+首先，进入 `vqaeval` 目录下，安装必要的依赖，并创建 `downloads` 子目录，用于存储下载的数据集：
+```bash
+cd vqaeval
+pip install -r requirements.txt
+mkdir downloads
+```
+<br />
+
+然后，从下列各地址下载数据集并置于指定目录下：
+###### TextVQA
+```bash
+cd downloads
+mkdir TextVQA && cd TextVQA
+wget https://dl.fbaipublicfiles.com/textvqa/images/train_val_images.zip
+unzip train_val_images.zip && rm train_val_images.zip
+mv train_val_images/train_images . && rm -rf train_val_images
+wget https://dl.fbaipublicfiles.com/textvqa/data/TextVQA_0.5.1_val.json
+cd ../..
+```
+
+###### DocVQA / DocVQATest
+```bash
+cd downloads
+mkdir DocVQA && cd DocVQA && mkdir spdocvqa_images
+# 在 https://rrc.cvc.uab.es/?ch=17&com=downloads 下载 Task 1 - Single Page Document Visual Question Answering 下的 Images 和 Annotations
+# 将下载得到的 spdocvqa_images.tar.gz 以及 spdocvqa_qas.zip 置于 DocVQA 目录下
+tar -zxvf spdocvqa_images.tar.gz -C spdocvqa_images && rm spdocvqa_images.tar.gz
+unzip spdocvqa_qas.zip && rm spdocvqa_qas.zip
+cp spdocvqa_qas/val_v1.0_withQT.json . && cp spdocvqa_qas/test_v1.0.json .  && rm -rf spdocvqa_qas
+cd ../..
+```
+<br />
+
+`downloads` 目录应当按照下列结构组织：
+```bash
+downloads
+├── TextVQA
+│   ├── train_images
+│   │   ├── ...
+│   ├── TextVQA_0.5.1_val.json
+├── DocVQA
+│   ├── spdocvqa_images
+│   │   ├── ...
+│   ├── val_v1.0_withQT.json
+│   ├── test_v1.0.json
+```
+<br />
+
+准备好相应的数据集之后，修改 `shell/run_inference.sh` 的参数，运行推理：
+
+```bash
+chmod +x ./shell/run_inference.sh
+./shell/run_inference.sh
+```
+<br />
+
+可以传入的参数位于 `eval_utils/getargs.py` 中，各主要参数的含义如下：
+```bash
+# 指定 TextVQA 评测所有图片和问题的路径
+--textVQA_image_dir
+--textVQA_ann_path
+# 指定 DocVQA 评测所有图片和问题的路径
+--docVQA_image_dir
+--docVQA_ann_path
+# 指定 DocVQATest 评测所有图片和问题的路径
+--docVQATest_image_dir
+--docVQATest_ann_path
+
+# 决定是否评测某个任务，eval_all 设置为 True 表示所有任务都评测
+--eval_textVQA
+--eval_docVQA
+--eval_docVQATest
+--eval_all
+
+# 模型名称、模型路径（从指定路径加载模型）
+--model_name
+--model_path
+# 从 checkpoint 加载模型
+--ckpt
+# 模型处理输入数据的方式，interleave 表示图文交错式，old 表示非交错式
+--generate_method
+# 推理时的批处理规模，建议推理时设置为 1
+--batchsize
+
+# 输出内容保存的路径
+--answer_path
+```
+<br />
+
+评测三个任务需要设置的参数如下：
+###### TextVQA
+```bash
+--eval_textVQA
+--textVQA_image_dir ./downloads/TextVQA/train_images
+--textVQA_ann_path ./downloads/TextVQA/TextVQA_0.5.1_val.json
+```
+
+###### DocVQA
+```bash
+--eval_docVQA
+--docVQA_image_dir ./downloads/DocVQA/spdocvqa_images
+--docVQA_ann_path ./downloads/DocVQA/val_v1.0_withQT.json
+```
+
+###### DocVQATest
+```bash
+--eval_docVQATest
+--docVQATest_image_dir ./downloads/DocVQA/spdocvqa_images
+--docVQATest_ann_path ./downloads/DocVQA/test_v1.0.json
+```
+<br />
+
+对于 DocVQATest 任务，为了将推理结果上传到[官方网站](https://rrc.cvc.uab.es/?ch=17)进行评测，还需要运行 `shell/run_transform.sh` 进行格式转换。其中，`input_file_path` 对应原始输出的 json 的路径，`output_file_path` 为自定义的转换后的 json 的路径：
+```bash
+chmod +x ./shell/run_transform.sh
+./shell/run_transform.sh
+```
\ No newline at end of file
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/requirements.txt b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/requirements.txt
new file mode 100644
index 0000000000000000000000000000000000000000..5bebbeec3f596a819e9340126d73f82a0d9d25d3
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/requirements.txt
@@ -0,0 +1,33 @@
+einops
+gradio==4.15.0
+huggingface_hub
+matplotlib
+numpy>=1.23.4
+omegaconf
+openai==1.3.5
+opencv-python>=4.4.0.46
+openpyxl
+pandas>=1.5.3
+pillow
+portalocker
+protobuf
+pycocoevalcap
+python-dotenv
+requests
+rich
+seaborn
+sentencepiece
+sty
+tabulate
+tiktoken
+timeout-decorator
+tqdm
+typing_extensions==4.7.1
+validators
+visual_genome
+xlsxwriter
+Pillow==10.1.0
+sentencepiece==0.1.99
+transformers==4.40.0
+torch==1.13.1
+torchvision
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/run.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/run.py
new file mode 100644
index 0000000000000000000000000000000000000000..c30cbd044fd975a39bc88cffff77ecc7d9ea5e53
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/run.py
@@ -0,0 +1,149 @@
+import torch
+import torch.distributed as dist
+from vlmeval.smp import *
+from vlmeval.evaluate import *
+from vlmeval.inference import infer_data_job
+from vlmeval.config import supported_VLM
+from vlmeval.utils import dataset_URLs, DATASET_TYPE, abbr2full, MMMU_result_transfer
+
+
+def parse_args():
+    parser = argparse.ArgumentParser()
+    parser.add_argument('--data', type=str, nargs='+', required=True)
+    parser.add_argument('--model', type=str, nargs='+', required=True)
+    parser.add_argument('--work-dir', type=str, default='.', help='select the output directory')
+    parser.add_argument('--mode', type=str, default='all', choices=['all', 'infer'])
+    parser.add_argument('--nproc', type=int, default=4, help='Parallel API calling')
+    parser.add_argument('--retry', type=int, default=None, help='retry numbers for API VLMs')
+    parser.add_argument('--judge', type=str, default=None)
+    parser.add_argument('--ignore', action='store_true', help='Ignore failed indices. ')
+    parser.add_argument('--verbose', action='store_true')
+    parser.add_argument('--rerun', action='store_true')
+    args = parser.parse_args()
+    return args
+
+
+def main():
+    logger = get_logger('RUN')
+
+    args = parse_args()
+    assert len(args.data), '--data should be a list of data files'
+
+    if args.retry is not None:
+        for k, v in supported_VLM.items():
+            if hasattr(v, 'keywords') and 'retry' in v.keywords:
+                v.keywords['retry'] = args.retry
+                supported_VLM[k] = v
+            if hasattr(v, 'keywords') and 'verbose' in v.keywords:
+                v.keywords['verbose'] = args.verbose
+                supported_VLM[k] = v
+
+    rank, world_size = get_rank_and_world_size()
+    if world_size > 1:
+        local_rank = os.environ.get('LOCAL_RANK', 0)
+        torch.cuda.set_device(int(local_rank))
+        dist.init_process_group(backend='nccl', timeout=datetime.timedelta(seconds=10800))
+
+    for _, model_name in enumerate(args.model):
+        model = None
+
+        pred_root = osp.join(args.work_dir, model_name)
+        os.makedirs(pred_root, exist_ok=True)
+
+        for _, dataset_name in enumerate(args.data):
+            custom_flag = False
+
+            if dataset_name not in dataset_URLs:
+                dataset_name = abbr2full(dataset_name)
+
+            if dataset_name not in dataset_URLs:
+                logger.warning(f'Dataset {dataset_name} is not officially supported. ')
+                file_path = osp.join(LMUDataRoot(), f'{dataset_name}.tsv')
+                if not osp.exists(file_path):
+                    logger.error(f'Cannot find the local dataset {dataset_name}. ')
+                    continue
+                else:
+                    custom_flag = True
+
+            result_file = f'{pred_root}/{model_name}_{dataset_name}.xlsx'
+            if osp.exists(result_file) and args.rerun:
+                os.system(f'rm {pred_root}/{model_name}_{dataset_name}_*')
+
+            if model is None:
+                model = model_name  # which is only a name
+
+            model = infer_data_job(
+                model,
+                work_dir=pred_root,
+                model_name=model_name,
+                dataset_name=dataset_name,
+                verbose=args.verbose,
+                api_nproc=args.nproc,
+                ignore_failed=args.ignore)
+
+            if rank == 0:
+                if dataset_name in ['MMMU_TEST']:
+                    result_json = MMMU_result_transfer(result_file)
+                    logger.info(f'Transfer MMMU_TEST result to json for official evaluation, json file saved in {result_json}')  # noqa: E501
+                    continue
+
+            if dataset_name in [
+                'MMBench_TEST_CN', 'MMBench_TEST_EN', 'MMBench', 'MMBench_CN'
+                'MMBench_TEST_CN_V11', 'MMBench_TEST_EN_V11', 'MMBench_V11', 'MMBench_CN_V11'
+            ]:
+                if not MMBenchOfficialServer(dataset_name):
+                    logger.error(
+                        f'Can not evaluate {dataset_name} on non-official servers, '
+                        'will skip the evaluation. '
+                    )
+                    continue
+
+            judge_kwargs = {
+                'nproc': args.nproc,
+                'verbose': args.verbose,
+            }
+            if args.retry is not None:
+                judge_kwargs['retry'] = args.retry
+            if args.judge is not None:
+                judge_kwargs['model'] = args.judge
+            else:
+                if DATASET_TYPE(dataset_name) in ['multi-choice', 'Y/N']:
+                    judge_kwargs['model'] = 'chatgpt-0613'
+                elif listinstr(['MMVet', 'MathVista', 'LLaVABench'], dataset_name):
+                    judge_kwargs['model'] = 'gpt-4-turbo'
+            if 'OPENAI_API_KEY_JUDGE' in os.environ and len(os.environ['OPENAI_API_KEY_JUDGE']):
+                judge_kwargs['key'] = os.environ['OPENAI_API_KEY_JUDGE']
+            if 'OPENAI_API_BASE_JUDGE' in os.environ and len(os.environ['OPENAI_API_BASE_JUDGE']):
+                judge_kwargs['api_base'] = os.environ['OPENAI_API_BASE_JUDGE']
+
+            if rank == 0 and args.mode == 'all':
+                if DATASET_TYPE(dataset_name) == 'multi-choice':
+                    dataset_name = 'default' if custom_flag else dataset_name
+                    multiple_choice_eval(
+                        result_file,
+                        dataset=dataset_name,
+                        **judge_kwargs)
+                elif DATASET_TYPE(dataset_name) == 'Y/N':
+                    YOrN_eval(
+                        result_file,
+                        dataset=dataset_name,
+                        **judge_kwargs)
+                elif DATASET_TYPE(dataset_name) == 'Caption':
+                    COCO_eval(result_file)
+                elif dataset_name == 'MMVet':
+                    MMVet_eval(result_file, **judge_kwargs)
+                elif dataset_name == 'OCRBench':
+                    OCRBench_eval(result_file)
+                elif listinstr(['OCRVQA', 'TextVQA', 'ChartQA', 'DocVQA', 'InfoVQA'], dataset_name):
+                    VQAEval(result_file, dataset_name)
+                elif listinstr(['MathVista'], dataset_name):
+                    MathVista_eval(result_file, **judge_kwargs)
+                elif listinstr(['LLaVABench'], dataset_name):
+                    LLaVABench_eval(result_file, **judge_kwargs)
+                else:
+                    logger.error(f'Dataset {dataset_name} is not handled by evaluator, will be skipped. ')
+
+
+if __name__ == '__main__':
+    load_env()
+    main()
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/script/run_inference.sh b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/script/run_inference.sh
new file mode 100644
index 0000000000000000000000000000000000000000..f9c3a5a5e99c701e35ac8d3995c2b36aa3aee6c3
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/script/run_inference.sh
@@ -0,0 +1,31 @@
+export PATH=/usr/local/cuda/bin:$PATH
+
+export HF_ENDPOINT=https://hf-mirror.com
+export OMP_NUM_THREADS=1
+export timestamp=`date +"%Y%m%d%H%M%S"`
+export OLD_VERSION='False'
+export PYTHONPATH=$(dirname $SELF_DIR):$PYTHONPATH
+
+# gpu consumed
+# fp16 17-18G
+# int4 7-8G
+
+# model to be used
+# Example: MODELNAME=MiniCPM-Llama3-V-2_5
+MODELNAME=$1
+# datasets to be tested
+# Example: DATALIST="POPE ScienceQA_TEST ChartQA_TEST"
+DATALIST=$2
+# test mode, all or infer
+MODE=$3
+
+echo "Starting inference with model $MODELNAME on datasets $DATALIST"
+# run on multi gpus with torchrun command
+# remember to run twice, the first run may fail
+torchrun --nproc_per_node=8 run.py --data $DATALIST --model $MODELNAME --mode $MODE
+torchrun --nproc_per_node=8 run.py --data $DATALIST --model $MODELNAME --mode $MODE
+# run on single gpu with python command
+# python run.py --data $DATALIST --model $MODELNAME --verbose --mode $MODE
+# python run.py --data $DATALIST --model $MODELNAME --verbose --mode $MODE
+
+ls
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/__init__.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..336a61a616f8051a193f87cf1548febd3903c7e7
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/__init__.py
@@ -0,0 +1,13 @@
+try:
+    import torch
+except ImportError:
+    pass
+
+from .smp import *
+from .api import *
+from .evaluate import *
+from .utils import *
+from .vlm import *
+from .config import *
+
+load_env()
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/api/__init__.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/api/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..050b1b666c67b6762cb3e33bb889a596a1acfe05
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/api/__init__.py
@@ -0,0 +1,6 @@
+from .gpt import OpenAIWrapper, GPT4V
+from .gpt_int import OpenAIWrapperInternal, GPT4V_Internal
+
+__all__ = [
+    'OpenAIWrapper', 'OpenAIWrapperInternal', 'GPT4V', 'GPT4V_Internal'
+]
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/api/base.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/api/base.py
new file mode 100644
index 0000000000000000000000000000000000000000..cd60c7d3d05fa2af352341538150c16e7a327d91
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/api/base.py
@@ -0,0 +1,195 @@
+import time
+import random as rd
+from abc import abstractmethod
+import os.path as osp
+import copy as cp
+from ..smp import get_logger, parse_file
+
+
+class BaseAPI:
+
+    allowed_types = ['text', 'image']
+    INTERLEAVE = True
+    INSTALL_REQ = False
+
+    def __init__(self,
+                 retry=10,
+                 wait=3,
+                 system_prompt=None,
+                 verbose=True,
+                 fail_msg='Failed to obtain answer via API.',
+                 **kwargs):
+        """Base Class for all APIs.
+
+        Args:
+            retry (int, optional): The retry times for `generate_inner`. Defaults to 10.
+            wait (int, optional): The wait time after each failed retry of `generate_inner`. Defaults to 3.
+            system_prompt (str, optional): Defaults to None.
+            verbose (bool, optional): Defaults to True.
+            fail_msg (str, optional): The message to return when failed to obtain answer.
+                Defaults to 'Failed to obtain answer via API.'.
+            **kwargs: Other kwargs for `generate_inner`.
+        """
+
+        self.wait = wait
+        self.retry = retry
+        self.system_prompt = system_prompt
+        self.verbose = verbose
+        self.fail_msg = fail_msg
+        self.logger = get_logger('ChatAPI')
+
+        if len(kwargs):
+            self.logger.info(f'BaseAPI received the following kwargs: {kwargs}')
+            self.logger.info('Will try to use them as kwargs for `generate`. ')
+        self.default_kwargs = kwargs
+
+    @abstractmethod
+    def generate_inner(self, inputs, **kwargs):
+        """The inner function to generate the answer.
+
+        Returns:
+            tuple(int, str, str): ret_code, response, log
+        """
+        self.logger.warning('For APIBase, generate_inner is an abstract method. ')
+        assert 0, 'generate_inner not defined'
+        ret_code, answer, log = None, None, None
+        # if ret_code is 0, means succeed
+        return ret_code, answer, log
+
+    def working(self):
+        """If the API model is working, return True, else return False.
+
+        Returns:
+            bool: If the API model is working, return True, else return False.
+        """
+        retry = 3
+        while retry > 0:
+            ret = self.generate('hello')
+            if ret is not None and ret != '' and self.fail_msg not in ret:
+                return True
+            retry -= 1
+        return False
+
+    def check_content(self, msgs):
+        """Check the content type of the input. Four types are allowed: str, dict, liststr, listdict.
+
+        Args:
+            msgs: Raw input messages.
+
+        Returns:
+            str: The message type.
+        """
+        if isinstance(msgs, str):
+            return 'str'
+        if isinstance(msgs, dict):
+            return 'dict'
+        if isinstance(msgs, list):
+            types = [self.check_content(m) for m in msgs]
+            if all(t == 'str' for t in types):
+                return 'liststr'
+            if all(t == 'dict' for t in types):
+                return 'listdict'
+        return 'unknown'
+
+    def preproc_content(self, inputs):
+        """Convert the raw input messages to a list of dicts.
+
+        Args:
+            inputs: raw input messages.
+
+        Returns:
+            list(dict): The preprocessed input messages. Will return None if failed to preprocess the input.
+        """
+        if self.check_content(inputs) == 'str':
+            return [dict(type='text', value=inputs)]
+        elif self.check_content(inputs) == 'dict':
+            assert 'type' in inputs and 'value' in inputs
+            return [inputs]
+        elif self.check_content(inputs) == 'liststr':
+            res = []
+            for s in inputs:
+                mime, pth = parse_file(s)
+                if mime is None or mime == 'unknown':
+                    res.append(dict(type='text', value=s))
+                else:
+                    res.append(dict(type=mime.split('/')[0], value=pth))
+            return res
+        elif self.check_content(inputs) == 'listdict':
+            for item in inputs:
+                assert 'type' in item and 'value' in item
+                mime, s = parse_file(item['value'])
+                if mime is None:
+                    assert item['type'] == 'text', item['value']
+                else:
+                    assert mime.split('/')[0] == item['type']
+                    item['value'] = s
+            return inputs
+        else:
+            return None
+
+    def generate(self, message, **kwargs1):
+        """The main function to generate the answer. Will call `generate_inner` with the preprocessed input messages.
+
+        Args:
+            message: raw input messages.
+
+        Returns:
+            str: The generated answer of the Failed Message if failed to obtain answer.
+        """
+        assert self.check_content(message) in ['str', 'dict', 'liststr', 'listdict'], f'Invalid input type: {message}'
+        message = self.preproc_content(message)
+        assert message is not None and self.check_content(message) == 'listdict'
+        for item in message:
+            assert item['type'] in self.allowed_types, f'Invalid input type: {item["type"]}'
+
+        # merge kwargs
+        kwargs = cp.deepcopy(self.default_kwargs)
+        kwargs.update(kwargs1)
+
+        answer = None
+        # a very small random delay [0s - 0.5s]
+        T = rd.random() * 0.5
+        time.sleep(T)
+
+        for i in range(self.retry):
+            try:
+                ret_code, answer, log = self.generate_inner(message, **kwargs)
+                if ret_code == 0 and self.fail_msg not in answer and answer != '':
+                    if self.verbose:
+                        print(answer)
+                    return answer
+                elif self.verbose:
+                    if not isinstance(log, str):
+                        try:
+                            log = log.text
+                        except:
+                            self.logger.warning(f'Failed to parse {log} as an http response. ')
+                    self.logger.info(f'RetCode: {ret_code}\nAnswer: {answer}\nLog: {log}')
+            except Exception as err:
+                if self.verbose:
+                    self.logger.error(f'An error occured during try {i}:')
+                    self.logger.error(err)
+            # delay before each retry
+            T = rd.random() * self.wait * 2
+            time.sleep(T)
+
+        return self.fail_msg if answer in ['', None] else answer
+
+    def message_to_promptimg(self, message):
+        assert not self.INTERLEAVE
+        model_name = self.__class__.__name__
+        import warnings
+        warnings.warn(
+            f'Model {model_name} does not support interleaved input. '
+            'Will use the first image and aggregated texts as prompt. ')
+        num_images = len([x for x in message if x['type'] == 'image'])
+        if num_images == 0:
+            prompt = '\n'.join([x['value'] for x in message if x['type'] == 'text'])
+            image = None
+        elif num_images == 1:
+            prompt = '\n'.join([x['value'] for x in message if x['type'] == 'text'])
+            image = [x['value'] for x in message if x['type'] == 'image'][0]
+        else:
+            prompt = '\n'.join([x['value'] if x['type'] == 'text' else '<image>' for x in message])
+            image = [x['value'] for x in message if x['type'] == 'image'][0]
+        return prompt, image
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/api/gpt.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/api/gpt.py
new file mode 100644
index 0000000000000000000000000000000000000000..a8f19838905472d817719467606000c7738676d2
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/api/gpt.py
@@ -0,0 +1,178 @@
+from ..smp import *
+import os
+import sys
+from .base import BaseAPI
+
+APIBASES = {
+    'OFFICIAL': 'https://api.openai.com/v1/chat/completions',
+}
+
+
+def GPT_context_window(model):
+    length_map = {
+        'gpt-4-1106-preview': 128000,
+        'gpt-4-vision-preview': 128000,
+        'gpt-4': 8192,
+        'gpt-4-32k': 32768,
+        'gpt-4-0613': 8192,
+        'gpt-4-32k-0613': 32768,
+        'gpt-3.5-turbo-1106': 16385,
+        'gpt-3.5-turbo': 4096,
+        'gpt-3.5-turbo-16k': 16385,
+        'gpt-3.5-turbo-instruct': 4096,
+        'gpt-3.5-turbo-0613': 4096,
+        'gpt-3.5-turbo-16k-0613': 16385,
+    }
+    if model in length_map:
+        return length_map[model]
+    else:
+        return 128000
+
+
+class OpenAIWrapper(BaseAPI):
+
+    is_api: bool = True
+
+    def __init__(self,
+                 model: str = 'gpt-3.5-turbo-0613',
+                 retry: int = 5,
+                 wait: int = 5,
+                 key: str = None,
+                 verbose: bool = True,
+                 system_prompt: str = None,
+                 temperature: float = 0,
+                 timeout: int = 60,
+                 api_base: str = None,
+                 max_tokens: int = 1024,
+                 img_size: int = 512,
+                 img_detail: str = 'low',
+                 **kwargs):
+
+        self.model = model
+        self.cur_idx = 0
+        self.fail_msg = 'Failed to obtain answer via API. '
+        self.max_tokens = max_tokens
+        self.temperature = temperature
+
+        if 'step-1v' in model:
+            env_key = os.environ.get('STEPAI_API_KEY', '')
+            if key is None:
+                key = env_key
+        else:
+            env_key = os.environ.get('OPENAI_API_KEY', '')
+            if key is None:
+                key = env_key
+            assert isinstance(key, str) and key.startswith('sk-'), (
+                f'Illegal openai_key {key}. '
+                'Please set the environment variable OPENAI_API_KEY to your openai key. '
+            )
+        self.key = key
+        assert img_size > 0 or img_size == -1
+        self.img_size = img_size
+        assert img_detail in ['high', 'low']
+        self.img_detail = img_detail
+        self.timeout = timeout
+
+        super().__init__(wait=wait, retry=retry, system_prompt=system_prompt, verbose=verbose, **kwargs)
+
+        if api_base is None:
+            if 'OPENAI_API_BASE' in os.environ and os.environ['OPENAI_API_BASE'] != '':
+                self.logger.error('Environment variable OPENAI_API_BASE is set. Will use it as api_base. ')
+                api_base = os.environ['OPENAI_API_BASE']
+            else:
+                api_base = 'OFFICIAL'
+
+        assert api_base is not None
+
+        if api_base in APIBASES:
+            self.api_base = APIBASES[api_base]
+        elif api_base.startswith('http'):
+            self.api_base = api_base
+        else:
+            self.logger.error('Unknown API Base. ')
+            sys.exit(-1)
+        self.logger.info(f'Using API Base: {self.api_base}; API Key: {self.key}')
+
+    # inputs can be a lvl-2 nested list: [content1, content2, content3, ...]
+    # content can be a string or a list of image & text
+    def prepare_inputs(self, inputs):
+        input_msgs = []
+        if self.system_prompt is not None:
+            input_msgs.append(dict(role='system', content=self.system_prompt))
+        has_images = np.sum([x['type'] == 'image' for x in inputs])
+        if has_images:
+            content_list = []
+            for msg in inputs:
+                if msg['type'] == 'text':
+                    content_list.append(dict(type='text', text=msg['value']))
+                elif msg['type'] == 'image':
+                    from PIL import Image
+                    img = Image.open(msg['value'])
+                    b64 = encode_image_to_base64(img, target_size=self.img_size)
+                    img_struct = dict(url=f'data:image/jpeg;base64,{b64}', detail=self.img_detail)
+                    content_list.append(dict(type='image_url', image_url=img_struct))
+            input_msgs.append(dict(role='user', content=content_list))
+        else:
+            assert all([x['type'] == 'text' for x in inputs])
+            text = '\n'.join([x['value'] for x in inputs])
+            input_msgs.append(dict(role='user', content=text))
+        return input_msgs
+
+    def generate_inner(self, inputs, **kwargs) -> str:
+        input_msgs = self.prepare_inputs(inputs)
+        temperature = kwargs.pop('temperature', self.temperature)
+        max_tokens = kwargs.pop('max_tokens', self.max_tokens)
+
+        context_window = GPT_context_window(self.model)
+        max_tokens = min(max_tokens, context_window - self.get_token_len(inputs))
+        if 0 < max_tokens <= 100:
+            self.logger.warning(
+                'Less than 100 tokens left, '
+                'may exceed the context window with some additional meta symbols. '
+            )
+        if max_tokens <= 0:
+            return 0, self.fail_msg + 'Input string longer than context window. ', 'Length Exceeded. '
+
+        headers = {'Content-Type': 'application/json', 'Authorization': f'Bearer {self.key}'}
+        payload = dict(
+            model=self.model,
+            messages=input_msgs,
+            max_tokens=max_tokens,
+            n=1,
+            temperature=temperature,
+            **kwargs)
+        response = requests.post(self.api_base, headers=headers, data=json.dumps(payload), timeout=self.timeout * 1.1)
+        ret_code = response.status_code
+        ret_code = 0 if (200 <= int(ret_code) < 300) else ret_code
+        answer = self.fail_msg
+        try:
+            resp_struct = json.loads(response.text)
+            answer = resp_struct['choices'][0]['message']['content'].strip()
+        except:
+            pass
+        return ret_code, answer, response
+
+    def get_token_len(self, inputs) -> int:
+        import tiktoken
+        try:
+            enc = tiktoken.encoding_for_model(self.model)
+        except:
+            enc = tiktoken.encoding_for_model('gpt-4')
+        assert isinstance(inputs, list)
+        tot = 0
+        for item in inputs:
+            if item['type'] == 'text':
+                tot += len(enc.encode(item['value']))
+            elif item['type'] == 'image':
+                tot += 85
+                if self.img_detail == 'high':
+                    img = Image.open(item['value'])
+                    npatch = np.ceil(img.size[0] / 512) * np.ceil(img.size[1] / 512)
+                    tot += npatch * 170
+        return tot
+
+
+class GPT4V(OpenAIWrapper):
+
+    def generate(self, message, dataset=None):
+        return super(GPT4V, self).generate(message)
\ No newline at end of file
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/api/gpt_int.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/api/gpt_int.py
new file mode 100644
index 0000000000000000000000000000000000000000..e5e941c47d184e21750f6b8d06412a2e72d99b04
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/api/gpt_int.py
@@ -0,0 +1,90 @@
+import json
+import warnings
+import requests
+from ..smp import *
+from .gpt import GPT_context_window, OpenAIWrapper
+
+url = 'http://ecs.sv.us.alles-apin.openxlab.org.cn/v1/openai/v2/text/chat'
+headers = {
+    'Content-Type': 'application/json'
+}
+
+
+class OpenAIWrapperInternal(OpenAIWrapper):
+
+    is_api: bool = True
+
+    def __init__(self,
+                 model: str = 'gpt-3.5-turbo-0613',
+                 retry: int = 5,
+                 wait: int = 3,
+                 verbose: bool = True,
+                 system_prompt: str = None,
+                 temperature: float = 0,
+                 timeout: int = 60,
+                 max_tokens: int = 1024,
+                 img_size: int = 512,
+                 img_detail: str = 'low',
+                 **kwargs):
+
+        self.model = model
+        if 'KEYS' in os.environ and osp.exists(os.environ['KEYS']):
+            keys = load(os.environ['KEYS'])
+            headers['alles-apin-token'] = keys.get('alles-apin-token', '')
+        elif 'ALLES' in os.environ:
+            headers['alles-apin-token'] = os.environ['ALLES']
+        self.headers = headers
+        self.temperature = temperature
+        self.timeout = timeout
+        self.max_tokens = max_tokens
+
+        assert img_size > 0 or img_size == -1
+        self.img_size = img_size
+        assert img_detail in ['high', 'low']
+        self.img_detail = img_detail
+
+        super(OpenAIWrapper, self).__init__(
+            wait=wait, retry=retry, system_prompt=system_prompt, verbose=verbose, **kwargs)
+
+    def generate_inner(self, inputs, **kwargs) -> str:
+        input_msgs = self.prepare_inputs(inputs)
+
+        temperature = kwargs.pop('temperature', self.temperature)
+        max_tokens = kwargs.pop('max_tokens', self.max_tokens)
+
+        # Held out 100 tokens as buffer
+        context_window = GPT_context_window(self.model)
+        max_tokens = min(max_tokens, context_window - self.get_token_len(inputs))
+        if 0 < max_tokens <= 100:
+            print('Less than 100 tokens left, may exceed the context window with some additional meta symbols. ')
+        if max_tokens <= 0:
+            return 0, self.fail_msg + 'Input string longer than context window. ', 'Length Exceeded. '
+
+        payload = dict(
+            model=self.model,
+            messages=input_msgs,
+            max_tokens=max_tokens,
+            n=1,
+            stop=None,
+            timeout=self.timeout,
+            temperature=temperature,
+            **kwargs)
+
+        response = requests.post(url, headers=headers, data=json.dumps(payload), timeout=self.timeout * 1.1)
+        ret_code = response.status_code
+        ret_code = 0 if (200 <= int(ret_code) < 300) else ret_code
+
+        answer = self.fail_msg
+        try:
+            resp_struct = json.loads(response.text)
+            assert resp_struct['msg'] == 'ok' and resp_struct['msgCode'] == '10000', resp_struct
+            answer = resp_struct['data']['choices'][0]['message']['content'].strip()
+        except:
+            pass
+        return ret_code, answer, response
+
+
+class GPT4V_Internal(OpenAIWrapperInternal):
+
+    def generate(self, message, dataset=None):
+        return super(GPT4V_Internal, self).generate(message)
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/config.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/config.py
new file mode 100644
index 0000000000000000000000000000000000000000..2701f305571a4fad6a784426ab989e3ad1690e5b
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/config.py
@@ -0,0 +1,19 @@
+from vlmeval.vlm import *
+from vlmeval.api import *
+from functools import partial
+
+ungrouped = {
+    'MiniCPM-V':partial(MiniCPM_V, model_path='openbmb/MiniCPM-V'),
+    'MiniCPM-V-2':partial(MiniCPM_V, model_path='openbmb/MiniCPM-V-2'),
+    'MiniCPM-Llama3-V-2_5':partial(MiniCPM_Llama3_V, model_path='openbmb/MiniCPM-Llama3-V-2_5'),
+}
+
+supported_VLM = {}
+
+model_groups = [
+    ungrouped
+]
+
+for grp in model_groups:
+    supported_VLM.update(grp)
+
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/OCRBench.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/OCRBench.py
new file mode 100644
index 0000000000000000000000000000000000000000..c37ad0872b8b92729001daf9db892dcfccbfffa8
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/OCRBench.py
@@ -0,0 +1,65 @@
+from vlmeval.smp import *
+
+
+def OCRBench_eval(eval_file):
+    OCRBench_score = {
+        'Regular Text Recognition': 0,
+        'Irregular Text Recognition': 0,
+        'Artistic Text Recognition': 0,
+        'Handwriting Recognition': 0,
+        'Digit String Recognition': 0,
+        'Non-Semantic Text Recognition': 0,
+        'Scene Text-centric VQA': 0,
+        'Doc-oriented VQA': 0,
+        'Key Information Extraction': 0,
+        'Handwritten Mathematical Expression Recognition': 0
+    }
+
+    logger = get_logger('Evaluation')
+
+    data = load(eval_file)
+    lt = len(data)
+    lines = [data.iloc[i] for i in range(lt)]
+    for i in tqdm(range(len(lines))):
+        line = lines[i]
+        predict = str(line['prediction'])
+        answers = eval(line['answer'])
+        category = line['category']
+        if category == 'Handwritten Mathematical Expression Recognition':
+            for j in range(len(answers)):
+                answer = answers[j].strip().replace('\n', ' ').replace(' ', '')
+                predict = predict.strip().replace('\n', ' ').replace(' ', '')
+                if answer in predict:
+                    OCRBench_score[category] += 1
+                    break
+        else:
+            for j in range(len(answers)):
+                answer = answers[j].lower().strip().replace('\n', ' ')
+                predict = predict.lower().strip().replace('\n', ' ')
+                if answer in predict:
+                    OCRBench_score[category] += 1
+                    break
+
+    final_score_dict = {}
+    final_score_dict['Text Recognition'] = (
+        OCRBench_score['Regular Text Recognition'] + OCRBench_score['Irregular Text Recognition']
+        + OCRBench_score['Artistic Text Recognition'] + OCRBench_score['Handwriting Recognition']
+        + OCRBench_score['Digit String Recognition'] + OCRBench_score['Non-Semantic Text Recognition']
+    )
+    final_score_dict['Scene Text-centric VQA'] = OCRBench_score['Scene Text-centric VQA']
+    final_score_dict['Doc-oriented VQA'] = OCRBench_score['Doc-oriented VQA']
+    final_score_dict['Key Information Extraction'] = OCRBench_score['Key Information Extraction']
+    final_score_dict['Handwritten Mathematical Expression Recognition'] = \
+        OCRBench_score['Handwritten Mathematical Expression Recognition']
+    final_score_dict['Final Score'] = (
+        final_score_dict['Text Recognition'] + final_score_dict['Scene Text-centric VQA']
+        + final_score_dict['Doc-oriented VQA'] + final_score_dict['Key Information Extraction']
+        + final_score_dict['Handwritten Mathematical Expression Recognition']
+    )
+    final_score_dict['Final Score Norm'] = float(final_score_dict['Final Score']) / 10
+    score_pth = eval_file.replace('.xlsx', '_score.json')
+    dump(final_score_dict, score_pth)
+    logger.info(f'OCRBench_eval successfully finished evaluating {eval_file}, results saved in {score_pth}')
+    logger.info('Score: ')
+    for key, value in final_score_dict.items():
+        logger.info('{}:{}'.format(key, value))
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/__init__.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..10248c4b66d2daf7d3a5603990ec4b35c1194368
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/__init__.py
@@ -0,0 +1,9 @@
+from .yes_or_no import default_rating, MME_rating, YOrN_eval
+from .mmvet_eval import MMVet_eval
+from .multiple_choice import multiple_choice_eval
+from .coco_eval import COCO_eval
+from .vqa_eval import VQAEval
+from .mathvista_eval import MathVista_eval
+from .llavabench import LLaVABench_eval
+from .misc import build_judge
+from .OCRBench import OCRBench_eval
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/coco_eval.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/coco_eval.py
new file mode 100644
index 0000000000000000000000000000000000000000..9ae8f4f85bf50b8ede6c03c4ccdc573d8303203e
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/coco_eval.py
@@ -0,0 +1,74 @@
+from vlmeval.smp import *
+from pycocoevalcap.bleu.bleu import Bleu
+from pycocoevalcap.rouge.rouge import Rouge
+from pycocoevalcap.cider.cider import Cider
+
+
+class COCO_Caption_Scorer():
+    def __init__(self, ref, gt):
+        self.ref = ref
+        self.gt = gt
+        print('setting up scorers...')
+        self.scorers = [
+            (Bleu(4), ['Bleu_1', 'Bleu_2', 'Bleu_3', 'Bleu_4']),
+            # (Meteor(), "METEOR"), # need java version 11.0.16+
+            (Rouge(), 'ROUGE_L'),
+            (Cider(), 'CIDEr'),
+            # (Spice(), "SPICE"), # need java version 11.0.16+
+        ]
+
+    def compute_scores(self):
+        total_scores = {}
+        for scorer, method in self.scorers:
+            print('computing %s score...' % (scorer.method()))
+            score, scores = scorer.compute_score(self.gt, self.ref)
+            if type(method) == list:
+                for sc, scs, m in zip(score, scores, method):
+                    print('%s: %0.3f' % (m, sc * 100))
+                total_scores['Bleu'] = [x * 100 for x in score]
+            else:
+                print('%s: %0.3f' % (method, score * 100))
+                total_scores[method] = score * 100
+
+        print('*****DONE*****')
+        for key, value in total_scores.items():
+            print('{}:{}'.format(key, value))
+        return total_scores
+
+
+def COCO_eval(eval_file, nproc=4, verbose=False):
+    logger = get_logger('Evaluation')
+
+    data = load(eval_file)
+
+    lt = len(data)
+    lines = [data.iloc[i] for i in range(lt)]
+    ref = {}
+    gt = {}
+    for i, line in enumerate(lines):
+        ref[str(i)] = [str(line['prediction'])]
+        gt[str(i)] = eval(line['answer'])
+
+    scorer = COCO_Caption_Scorer(ref, gt)
+    coco_caption_score_dict = scorer.compute_scores()
+
+    score_pth = eval_file.replace('.xlsx', '_score.json')
+    dump(coco_caption_score_dict, score_pth)
+    logger.info(f'COCO_eval successfully finished evaluating {eval_file}, results saved in {score_pth}')
+    logger.info('Score: ')
+    for key, value in coco_caption_score_dict.items():
+        logger.info('{}:{}'.format(key, value))
+
+
+def parse_args():
+    parser = argparse.ArgumentParser(description='Inference LLM Answers. ')
+    parser.add_argument('--data', type=str, help='The question set for inference, in excel / tsv / json format. ')
+    parser.add_argument('--nproc', type=int, default=4)
+    parser.add_argument('--verbose', action='store_true')
+    args = parser.parse_args()
+    return args
+
+
+if __name__ == '__main__':
+    args = parse_args()
+    COCO_eval(eval_file=args.data, nproc=args.nproc, verbose=args.verbose)
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/llavabench.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/llavabench.py
new file mode 100644
index 0000000000000000000000000000000000000000..1eac9cce0d1405427b00cb39336ae4e7091eb7df
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/llavabench.py
@@ -0,0 +1,120 @@
+import argparse
+import numpy as np
+import pandas as pd
+import os.path as osp
+from vlmeval.evaluate.misc import build_judge
+from vlmeval.smp import *
+from vlmeval.utils import track_progress_rich
+
+rule_dict = {
+    'llava_bench_conv': {'role': 'Assistant', 'prompt': 'We would like to request your feedback on the performance of two AI assistants in response to the user question displayed above. The user asks the question on observing an image. For your reference, the visual content in the image is represented with a few sentences describing the image. \nPlease rate the helpfulness, relevance, accuracy, level of details of their responses. Each assistant receives an overall score on a scale of 1 to 10, where a higher score indicates better overall performance.\nPlease first output a single line containing only two values indicating the scores for Assistant 1 and 2, respectively. The two scores are separated by a space.\nIn the subsequent line, please provide a comprehensive explanation of your evaluation, avoiding any potential bias and ensuring that the order in which the responses were presented does not affect your judgment.'},  # noqa: E501
+    'llava_bench_detail': {'role': 'Assistant', 'prompt': 'We would like to request your feedback on the performance of two AI assistants in response to the user question displayed above. The user asks the question on observing an image. For your reference, the visual content in the image is represented with a few sentences describing the image. \nPlease rate the helpfulness, relevance, accuracy, level of details of their responses. Each assistant receives an overall score on a scale of 1 to 10, where a higher score indicates better overall performance.\nPlease first output a single line containing only two values indicating the scores for Assistant 1 and 2, respectively. The two scores are separated by a space.\nIn the subsequent line, please provide a comprehensive explanation of your evaluation, avoiding any potential bias and ensuring that the order in which the responses were presented does not affect your judgment.'},  # noqa: E501
+    'llava_bench_complex': {'role': 'Assistant', 'prompt': 'We would like to request your feedback on the performance of two AI assistants in response to the user question displayed above. The user asks the question on observing an image. For your reference, the visual content in the image is represented with a few sentences describing the image. \nPlease rate the helpfulness, relevance, accuracy, level of details of their responses. Each assistant receives an overall score on a scale of 1 to 10, where a higher score indicates better overall performance.\nPlease first output a single line containing only two values indicating the scores for Assistant 1 and 2, respectively. The two scores are separated by a space.\nIn the subsequent line, please provide a comprehensive explanation of your evaluation, avoiding any potential bias and ensuring that the order in which the responses were presented does not affect your judgment.'}  # noqa: E501
+}
+
+
+def get_eval(judge, content):
+    return judge.generate(content)
+
+
+def parse_score(review):
+    logger = get_logger('Evaluation')
+    try:
+        score_pair = review.split('\n')[0]
+        score_pair = score_pair.replace(',', ' ')
+        sp = score_pair.split(' ')
+        if len(sp) == 2:
+            return [float(sp[0]), float(sp[1])]
+        else:
+            logger.error('error', review)
+            return [-1, -1]
+    except Exception as e:
+        logger.error(e, 'error', review)
+        return [-1, -1]
+
+
+def build_prompt(line):
+    cap_str = line['caption']
+    question = line['question']
+    ans1 = line['gpt4_ans']
+    ans2 = line['prediction']
+    category = 'llava_bench_' + line['category']
+    rule = rule_dict[category]
+    role, prompt = rule['role'], rule['prompt']
+
+    content = (f'[Context]\n{cap_str}\n\n'
+               f'[Question]\n{question}\n\n'
+               f'[{role} 1]\n{ans1}\n\n[End of {role} 1]\n\n'
+               f'[{role} 2]\n{ans2}\n\n[End of {role} 2]\n\n'
+               f'[System]\n{prompt}\n\n')
+    return content
+
+
+def LLaVABench_atomeval(model, prompt):
+    review = get_eval(model, prompt)
+    scores = parse_score(review)
+    return scores
+
+
+def LLaVABench_score(data):
+    cates = ['overall'] + list(set(data['category']))
+    ret = defaultdict(list)
+
+    for c in cates:
+        ret['split'].append(c)
+        sub = data[data['category'] == c] if c != 'overall' else data
+        ret['Relative Score (main)'].append(np.mean(sub['score']) / np.mean(sub['gpt4_score']) * 100)
+        ret['VLM Score'].append(np.mean(sub['score']) * 10)
+        ret['GPT4 Score'].append(np.mean(sub['gpt4_score']) * 10)
+    return pd.DataFrame(ret)
+
+
+def LLaVABench_eval(eval_file, **judge_kwargs):
+    suffix = '.' + eval_file.split('.')[-1]
+    record_file = eval_file.replace(suffix, '_openai_result' + suffix)
+    score_file = eval_file.replace(suffix, '_score.csv')
+    nproc = judge_kwargs.pop('nproc', 4)
+
+    if not osp.exists(record_file):
+        data = load(eval_file)
+        lines = [data.iloc[i] for i in range(len(data))]
+        model = build_judge(
+            temperature=0.2,
+            system_prompt='You are a helpful and precise assistant for checking the quality of the answer.',
+            **judge_kwargs)
+        prompts = [build_prompt(line) for line in lines]
+        tups = [(model, prompt) for prompt in prompts]
+        scores = track_progress_rich(LLaVABench_atomeval, tups, nproc=nproc, chunksize=nproc)
+        data['gpt4_score'] = [x[0] for x in scores]
+        data['score'] = [x[1] for x in scores]
+        dump(data, record_file)
+
+    data = load(record_file)
+    ret = LLaVABench_score(data).round(1)
+    print(ret)
+    dump(ret, score_file)
+    return ret
+
+
+def parse_args():
+    parser = argparse.ArgumentParser(description='LLaVABench Evaluation. ')
+    parser.add_argument('data', type=str, help='The question set for inference, in excel / tsv / json format. ')
+    parser.add_argument(
+        '--model', type=str, help='The LLM (GPT) used for inference. ', default='gpt-4-turbo',
+        choices=['gpt-4-0613', 'gpt-4-turbo', 'chatgpt-1106', 'chatgpt-0613', 'gpt-4-0314'])
+    parser.add_argument('--nproc', type=int, default=4)
+    parser.add_argument('--verbose', action='store_true')
+    args = parser.parse_args()
+    return args
+
+
+if __name__ == '__main__':
+    load_env()
+    args = parse_args()
+    judge_kwargs = dict(model=args.model, nproc=args.nproc, verbose=args.verbose)
+    if 'OPENAI_API_KEY_JUDGE' in os.environ and os.environ['OPENAI_API_KEY_JUDGE']:
+        judge_kwargs['key'] = os.environ['OPENAI_API_KEY_JUDGE']
+    if 'OPENAI_API_BASE_JUDGE' in os.environ and os.environ['OPENAI_API_BASE_JUDGE']:
+        judge_kwargs['api_base'] = os.environ['OPENAI_API_BASE_JUDGE']
+
+    LLaVABench_eval(eval_file=args.data, **judge_kwargs)
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/mathvista_eval.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/mathvista_eval.py
new file mode 100644
index 0000000000000000000000000000000000000000..67c5e8b2981ae8f19e7d143d54f5268d4e864a5a
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/mathvista_eval.py
@@ -0,0 +1,240 @@
+from vlmeval.evaluate.misc import build_judge
+from vlmeval.smp import *
+from vlmeval.utils import track_progress_rich
+from vlmeval.utils.matching_util import can_infer
+
+
+def get_gpt4_ICE():
+    example_1 = """
+Hint: Please answer the question requiring an integer answer and provide the final value,
+e.g., 1, 2, 3, at the end.\n
+Question: Which number is missing?\n
+Model response: The number missing in the sequence is 14.\n
+Extracted answer: 14
+"""
+
+    example_2 = """
+Hint: Please answer the question requiring a floating-point number with one decimal place and provide the final value,
+e.g., 1.2, 1.3, 1.4, at the end.\n
+Question: What is the fraction of females facing the camera?\n
+Model response: The fraction of females facing the camera is 0.6,
+which means that six out of ten females in the group are facing the camera.\n
+Extracted answer: 0.6
+"""
+
+    example_3 = """
+Hint: Please answer the question requiring a floating-point number with two decimal places and provide the final value,
+e.g., 1.23, 1.34, 1.45, at the end.\n
+Question: How much money does Luca need to buy a sour apple candy and a butter-scotch candy? (Unit: $)\n
+Model response: Luca needs $1.45 to buy a sour apple candy and a butterscotch candy.\n
+Extracted answer: 1.45
+"""
+
+    example_4 = """
+Hint: Please answer the question requiring a Python list as an answer and provide the final list,
+e.g., [1, 2, 3], [1.2, 1.3, 1.4], at the end.\n
+Question: Between which two years does the line graph saw its maximum peak?\n
+Model response: The line graph saw its maximum peak between 2007 and 2008.\n
+Extracted answer: [2007, 2008]
+"""
+
+    example_5 = """
+Hint: Please answer the question and provide the correct option letter, e.g., A, B, C, D, at the end.\n
+Question: What fraction of the shape is blue?\n
+Choices: (A) 3/11 (B) 8/11 (C) 6/11 (D) 3/5\n
+Model response: The correct answer is (B) 8/11.\n
+Extracted answer: B
+"""
+
+    return [example_1, example_2, example_3, example_4, example_5]
+
+
+def build_mathvista_gpt4_prompt(line):
+    task_description = """
+Please read the following example.
+Then extract the answer from the model response and type it at the end of the prompt.\n
+"""
+    question = line['question']
+    prediction = str(line['prediction'])
+    prompt = task_description
+    examples = get_gpt4_ICE()
+    for example in examples:
+        prompt += example + '\n'
+    prompt += question + '\n'
+    prompt += 'Model respone: ' + prediction
+    prompt += 'Extracted answer:'
+    return prompt
+
+
+def list_to_dict(lst):
+    return {chr(65 + i): val for i, val in enumerate(lst)}
+
+
+def post_check(line, prefetch=False):
+    res = None
+    ans = line['answer']
+    response = line['prediction'] if prefetch else line['res']
+    try:
+        if line['question_type'] == 'multi_choice':
+            ans = line['answer_option']
+            choices = list_to_dict(eval(line['choices']))
+            res = can_infer(response, choices)
+            if prefetch:
+                return res
+        else:
+            if line['answer_type'] == 'integer':
+                res = int(response)
+                ans = int(line['answer'])
+            elif line['answer_type'] == 'float':
+                res = float(response)
+                ans = float(line['answer'])
+            else:
+                res = str(res)
+                ans = str(ans)
+    except ValueError:
+        pass
+
+    if res == ans:
+        return res if prefetch else True
+    else:
+        return False
+
+
+def MathVista_auxeval(model, line):
+    prompt = build_mathvista_gpt4_prompt(line)
+    log = ''
+    retry = 5
+    if post_check(line, prefetch=True):
+        res = post_check(line, prefetch=True)
+        return dict(log='Prefetch succeed', res=res)
+    for i in range(retry):
+        prediction = line['prediction']
+        res = model.generate(prompt, temperature=i * 0.5)
+        if res is None:
+            log += f'Try {i}: output is {prediction}, failed to parse.\n'
+        else:
+            log += 'Succeed'
+            return dict(log=log, res=res)
+    log += 'All 5 retries failed.\n'
+    return dict(log=log, res='')
+
+
+def MathVista_acc(result_file):
+    data = load(result_file)
+    tot = defaultdict(lambda: 0)
+    fetch = defaultdict(lambda: 0)
+    hit = defaultdict(lambda: 0)
+    lt = len(data)
+    skill_list = []
+    for i in range(lt):
+        item = data.iloc[i]
+        cate = item['task']
+        tot['Overall'] += 1
+        try:
+            skills = eval(item['skills'])
+        except SyntaxError:
+            skills = [item['skills']]
+        for skill in skills:
+            if skill not in skill_list:
+                skill_list.append(skill)
+            tot[skill] += 1
+        tot[cate] += 1
+        if item['log'] == 'Prefetch succeed':
+            fetch['Overall'] += 1
+            fetch[cate] += 1
+            for skill in skills:
+                fetch[skill] += 1
+        if post_check(item, prefetch=False):
+            hit['Overall'] += 1
+            hit[cate] += 1
+            for skill in skills:
+                hit[skill] += 1
+
+    res = defaultdict(list)
+    for k in tot.keys():
+        res['Task&Skill'].append(k)
+        res['tot'].append(tot[k])
+        res['prefetch'].append(fetch[k])
+        res['hit'].append(hit[k])
+        res['prefetch_rate'].append(fetch[k] / tot[k] * 100)
+        res['acc'].append(hit[k] / tot[k] * 100)
+    res = pd.DataFrame(res)
+    return res
+
+
+def MathVista_eval(eval_file, **judge_kwargs):
+    logger = get_logger('Evaluation')
+    model = judge_kwargs['model']
+
+    suffix = eval_file.split('.')[-1]
+    storage = eval_file.replace(f'.{suffix}', f'_{model}.xlsx')
+    tmp_file = eval_file.replace(f'.{suffix}', f'_{model}.pkl')
+    nproc = judge_kwargs.pop('nproc', 4)
+
+    if osp.exists(storage):
+        logger.warning(f'GPT scoring file {storage} already exists, will reuse it in MathVista_eval. ')
+    else:
+        data = load(eval_file)
+        model = build_judge(max_tokens=128, **judge_kwargs)
+        lt = len(data)
+        lines = [data.iloc[i] for i in range(lt)]
+        tups = [(model, line) for line in lines]
+        indices = [line['index'] for line in lines]
+
+        ans = {}
+        if osp.exists(tmp_file):
+            ans = load(tmp_file)
+        tups = [x for x, i in zip(tups, indices) if i not in ans]
+        indices = [i for i in indices if i not in ans]
+
+        if len(indices):
+            new_results = track_progress_rich(
+                MathVista_auxeval, tups, nproc=nproc, chunksize=nproc,
+                keys=indices, save=tmp_file)
+            ans = load(tmp_file)
+            for k, v in zip(indices, new_results):
+                assert k in ans
+                assert ans[k]['log'] == v['log'] and ans[k]['res'] == v['res']
+
+        log_map, res_map = {}, {}
+        all_inds = [line['index'] for line in lines]
+        for k in all_inds:
+            log_map[k] = ans[k]['log']
+            res_map[k] = ans[k]['res']
+        data['res'] = [res_map[idx] for idx in data['index']]
+        data['log'] = [log_map[idx] for idx in data['index']]
+        dump(data, storage)
+
+    score = MathVista_acc(storage)
+    score_pth = storage.replace('.xlsx', '_score.csv')
+
+    dump(score, score_pth)
+    logger.info(f'MathVista_eval successfully finished evaluating {eval_file}, results saved in {score_pth}')
+    logger.info('Score: ')
+    logger.info(score)
+
+
+def parse_args():
+    parser = argparse.ArgumentParser(description='Inference LLM Answers. ')
+    parser.add_argument('data', type=str, help='The question set for inference, in excel / tsv / json format. ')
+    parser.add_argument(
+        '--model',
+        type=str,
+        help='The LLM (GPT) used for inference. ',
+        default='gpt-4-turbo',
+        choices=['gpt-4-0613', 'gpt-4-turbo', 'chatgpt-1106', 'chatgpt-0613'])
+    parser.add_argument('--nproc', type=int, default=4)
+    parser.add_argument('--verbose', action='store_true')
+    args = parser.parse_args()
+    return args
+
+
+if __name__ == '__main__':
+    load_env()
+    args = parse_args()
+    judge_kwargs = dict(model=args.model, nproc=args.nproc, verbose=args.verbose)
+    if 'OPENAI_API_KEY_JUDGE' in os.environ and os.environ['OPENAI_API_KEY_JUDGE']:
+        judge_kwargs['key'] = os.environ['OPENAI_API_KEY_JUDGE']
+    if 'OPENAI_API_BASE_JUDGE' in os.environ and os.environ['OPENAI_API_BASE_JUDGE']:
+        judge_kwargs['api_base'] = os.environ['OPENAI_API_BASE_JUDGE']
+    MathVista_eval(eval_file=args.data, **judge_kwargs)
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/misc.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/misc.py
new file mode 100644
index 0000000000000000000000000000000000000000..ddf0c9e01843e04faeda43d93ebfffe901e2028f
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/misc.py
@@ -0,0 +1,29 @@
+import os
+from vlmeval.api import OpenAIWrapper, OpenAIWrapperInternal
+from vlmeval.smp import load_env
+
+INTERNAL = os.environ.get('INTERNAL', 0)
+
+
+def build_judge(**kwargs):
+    model = kwargs.pop('model', None)
+    load_env()
+    LOCAL_LLM = os.environ.get('LOCAL_LLM', None)
+    if LOCAL_LLM is None:
+        model_map = {
+            'gpt-4-turbo': 'gpt-4-1106-preview',
+            'gpt-4-0613': 'gpt-4-0613',
+            'gpt-4-0314': 'gpt-4-0314',
+            'gpt-4-0125': 'gpt-4-0125-preview',
+            'chatgpt-1106': 'gpt-3.5-turbo-1106',
+            'chatgpt-0613': 'gpt-3.5-turbo-0613',
+            'chatgpt-0125': 'gpt-3.5-turbo-0125'
+        }
+        model_version = model_map[model]
+    else:
+        model_version = LOCAL_LLM
+    if INTERNAL:
+        model = OpenAIWrapperInternal(model_version, **kwargs)
+    else:
+        model = OpenAIWrapper(model_version, **kwargs)
+    return model
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/mmvet_eval.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/mmvet_eval.py
new file mode 100644
index 0000000000000000000000000000000000000000..b0067e7c35015642249b91b21ac984f37a2028c0
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/mmvet_eval.py
@@ -0,0 +1,191 @@
+from vlmeval.evaluate.misc import build_judge
+from vlmeval.smp import *
+from vlmeval.utils import track_progress_rich
+
+
+def build_mmvet_gpt4_prompt(line):
+    question = line['question']
+    gt = str(line['answer'])
+    prediction = str(line['prediction'])
+    prompt = """
+Compare the ground truth and prediction from AI models, to give a correctness score for the prediction.
+<AND> in the ground truth means it is totally right
+only when all elements in the ground truth are present in the prediction,
+and <OR> means it is totally right when any one element in the ground truth is present in the prediction.
+The correctness score is 0.0 (totally wrong), 0.1, 0.2, 0.3, 0.4, 0.5, 0.6, 0.7, 0.8, 0.9, or 1.0 (totally right).
+Just complete the last space of the correctness score.
+
+Question | Ground truth | Prediction | Correctness
+--- | --- | --- | ---
+What is x in the equation? | -1 <AND> -5 | x = 3 | 0.0
+What is x in the equation? | -1 <AND> -5 | x = -1 | 0.5
+What is x in the equation? | -1 <AND> -5 | x = -5 | 0.5
+What is x in the equation? | -1 <AND> -5 | x = -5 or 5 | 0.5
+What is x in the equation? | -1 <AND> -5 | x = -1 or x = -5 | 1.0
+Can you explain this meme? | This meme is poking fun at the fact that the names of the countries
+Iceland and Greenland are misleading. Despite its name, Iceland is known for its beautiful green landscapes,
+while Greenland is mostly covered in ice and snow. The meme is saying that the person has trust issues
+because the names of these countries do not accurately represent their landscapes. |
+The meme talks about Iceland and Greenland. It's pointing out that despite their names,
+Iceland is not very icy and Greenland isn't very green. | 0.4
+Can you explain this meme? | This meme is poking fun at the fact that the names of the countries
+Iceland and Greenland are misleading. Despite its name, Iceland is known for its beautiful green landscapes,
+while Greenland is mostly covered in ice and snow. The meme is saying that the person has trust issues
+because the names of these countries do not accurately represent their landscapes. |
+The meme is using humor to point out the misleading nature of Iceland's and Greenland's names.
+Iceland, despite its name, has lush green landscapes while Greenland is mostly covered in ice and snow.
+The text 'This is why I have trust issues' is a playful way to suggest
+that these contradictions can lead to distrust or confusion.
+The humor in this meme is derived from the unexpected contrast between the names of the countries
+and their actual physical characteristics. | 1.0
+"""
+    gpt4_prompt = prompt + '\n' + ' | '.join(
+        [question, gt.replace('<AND>', ' <AND> ').replace('<OR>', ' <OR> '), prediction, ''])
+    return gpt4_prompt
+
+
+def MMVet_auxeval(model, line):
+    def float_cvt(s):
+        try:
+            return float(s)
+        except ValueError:
+            return None
+
+    prompt = build_mmvet_gpt4_prompt(line)
+    log = ''
+    retry = 5
+    for i in range(retry):
+        output = model.generate(prompt, temperature=i * 0.5)
+        score = float_cvt(output)
+        if score is None:
+            log += f'Try {i}: output is {output}, failed to parse.\n'
+        elif score < 0 or score > 1:
+            log += f'Try {i}: output is {output}, invalid score: {score}.\n'
+        else:
+            log += 'Succeed'
+            return dict(log=log, score=score)
+    log += 'All 5 retries failed.\n'
+    return dict(log=log, score=0.0)
+
+
+def MMVet_acc(result_file):
+    data = load(result_file)
+    tot = defaultdict(lambda: 0)
+    score = defaultdict(lambda: 0)
+    lt = len(data)
+    cate2_list = []
+    for i in range(lt):
+        item = data.iloc[i]
+        cate = item['category']
+        cate2 = cate.replace(',', '_')
+        if cate2 not in cate2_list:
+            cate2_list.append(cate2)
+        grade = float(item['score'])
+        cate_list = ['rec', 'ocr', 'know', 'gen', 'spat', 'math']
+        for capa in cate_list:
+            if capa in cate:
+                tot[capa] += 1
+                score[capa] += grade
+        tot['Overall'] += 1
+        tot[cate2] += 1
+        score['Overall'] += grade
+        score[cate2] += grade
+
+    res = defaultdict(list)
+    res2 = defaultdict(list)
+    cate_list.append('Overall')
+    cate2_list.append('Overall')
+    for k in cate_list:
+        res['Category'].append(k)
+        res['tot'].append(tot[k])
+        res['acc'].append(score[k] / tot[k] * 100)
+    for v in cate2_list:
+        res2['Category'].append(v)
+        res2['tot'].append(tot[v])
+        res2['acc'].append(score[v] / tot[v] * 100)
+    res = pd.DataFrame(res)
+    res2 = pd.DataFrame(res2)
+    return res, res2
+
+
+def MMVet_eval(eval_file, **judge_kwargs):
+    logger = get_logger('Evaluation')
+
+    suffix = eval_file.split('.')[-1]
+    model = judge_kwargs['model']
+    storage = eval_file.replace(f'.{suffix}', f'_{model}.xlsx')
+    tmp_file = eval_file.replace(f'.{suffix}', f'_{model}.pkl')
+    nproc = judge_kwargs.pop('nproc', 4)
+    if osp.exists(storage):
+        logger.warning(f'GPT scoring file {storage} already exists, will reuse it in MMVet_eval. ')
+    else:
+        data = load(eval_file)
+        model = build_judge(max_tokens=3, **judge_kwargs)
+
+        lt = len(data)
+        lines = [data.iloc[i] for i in range(lt)]
+        tups = [(model, line) for line in lines]
+        indices = [line['index'] for line in lines]
+
+        ans = {}
+        if osp.exists(tmp_file):
+            ans = load(tmp_file)
+        tups = [x for x, i in zip(tups, indices) if i not in ans]
+        indices = [i for i in indices if i not in ans]
+
+        if len(indices):
+            new_results = track_progress_rich(
+                MMVet_auxeval, tups, nproc=nproc, chunksize=nproc,
+                keys=indices, save=tmp_file)
+            ans = load(tmp_file)
+            for k, v in zip(indices, new_results):
+                assert k in ans
+                assert ans[k]['log'] == v['log'] and ans[k]['score'] == v['score']
+
+        log_map, score_map = {}, {}
+        all_inds = [line['index'] for line in lines]
+        for k in all_inds:
+            log_map[k] = ans[k]['log']
+            score_map[k] = ans[k]['score']
+        data['score'] = [score_map[idx] for idx in data['index']]
+        data['log'] = [log_map[idx] for idx in data['index']]
+        dump(data, storage)
+
+    score, score_fine = MMVet_acc(storage)
+    score_pth = storage.replace('.xlsx', '_score.csv')
+    score_fine_pth = storage.replace('.xlsx', '_score_fine.csv')
+
+    dump(score, score_pth)
+    dump(score_fine, score_fine_pth)
+    logger.info(
+        f'MMVet_eval successfully finished evaluating {eval_file}, '
+        f'results saved in {score_pth} and {score_fine_pth}'
+    )
+    logger.info('Score: ')
+    logger.info(score)
+
+
+def parse_args():
+    parser = argparse.ArgumentParser(description='Inference LLM Answers. ')
+    parser.add_argument('data', type=str, help='The question set for inference, in excel / tsv / json format. ')
+    parser.add_argument(
+        '--model',
+        type=str,
+        help='The LLM (GPT) used for inference. ',
+        default='gpt-4-turbo',
+        choices=['gpt-4-0613', 'gpt-4-turbo', 'chatgpt-1106', 'chatgpt-0613'])
+    parser.add_argument('--nproc', type=int, default=4)
+    parser.add_argument('--verbose', action='store_true')
+    args = parser.parse_args()
+    return args
+
+
+if __name__ == '__main__':
+    load_env()
+    args = parse_args()
+    judge_kwargs = dict(model=args.model, nproc=args.nproc, verbose=args.verbose)
+    if 'OPENAI_API_KEY_JUDGE' in os.environ and os.environ['OPENAI_API_KEY_JUDGE']:
+        judge_kwargs['key'] = os.environ['OPENAI_API_KEY_JUDGE']
+    if 'OPENAI_API_BASE_JUDGE' in os.environ and os.environ['OPENAI_API_BASE_JUDGE']:
+        judge_kwargs['api_base'] = os.environ['OPENAI_API_BASE_JUDGE']
+    MMVet_eval(eval_file=args.data, **judge_kwargs)
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/multiple_choice.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/multiple_choice.py
new file mode 100644
index 0000000000000000000000000000000000000000..50d52bf2edb3e42fd3ebd0d91730fa6dd553969f
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/multiple_choice.py
@@ -0,0 +1,399 @@
+import os.path as osp
+import pandas as pd
+from tqdm import tqdm
+from vlmeval.evaluate.misc import build_judge
+from vlmeval.utils import can_infer, track_progress_rich, TSVDataset
+from vlmeval.smp import *
+import numpy as np
+
+INTERNAL = os.environ.get('INTERNAL', 0)
+
+abbrs = {
+    'coarse_perception': 'CP',
+    'finegrained_perception (instance-level)': 'FP-S',
+    'finegrained_perception (cross-instance)': 'FP-C',
+    'logic_reasoning': 'LR',
+    'relation_reasoning': 'RR',
+    'attribute_reasoning': 'AR'
+}
+
+
+def MMMU_preproc(data):
+    logger = get_logger('Evaluation')
+    cnt = 0
+    As, Bs, Ans = list(data['A']), list(data['B']), list(data['answer'])
+    lt = len(data)
+    for i in range(lt):
+        if pd.isna(As[i]):
+            As[i] = Ans[i]
+            Bs[i] = 'Other Answers'
+            cnt += 1
+    logger.info(f'During MMMU_preproc in Evaluation, {cnt} open questions are re-formulated to multi-choice ones. ')
+    data['A'] = As
+    data['B'] = Bs
+    return data
+
+
+def report_acc(df):
+    # assert group in [None, 'category', 'l2-category']
+    res = defaultdict(list)
+
+    if 'split' in df:
+        splits = list(set(df['split']))
+        res['split'] = splits
+    else:
+        df['split'] = ['none'] * len(df)
+        res['split'] = ['none']
+
+    for group in [None, 'l2-category', 'category']:
+        if group is None:
+            res['Overall'] = [np.mean(df[df['split'] == sp]['hit']) for sp in res['split']]
+        elif group not in df:
+            continue
+        else:
+            abilities = list(set(df[group]))
+            abilities.sort()
+            for ab in abilities:
+                ab_name = abbrs[ab] if ab in abbrs else ab
+                sub_df = df[df[group] == ab]
+                res[ab_name] = [np.mean(sub_df[sub_df['split'] == sp]['hit']) for sp in res['split']]
+    return pd.DataFrame(res)
+
+
+def build_prompt(question, options, prediction):
+    tmpl = (
+        'You are an AI assistant who will help me to match '
+        'an answer with several options of a single-choice question. '
+        'You are provided with a question, several options, and an answer, '
+        'and you need to find which option is most similar to the answer. '
+        'If the meaning of all options are significantly different from the answer, output Z. '
+        'Your should output a single uppercase character in A, B, C, D (if they are valid options), and Z. \n'
+        'Example 1: \n'
+        'Question: What is the main object in image?\nOptions: A. teddy bear B. rabbit C. cat D. dog\n'
+        'Answer: a cute teddy bear\nYour output: A\n'
+        'Example 2: \n'
+        'Question: What is the main object in image?\nOptions: A. teddy bear B. rabbit C. cat D. dog\n'
+        'Answer: Spider\nYour output: Z\n'
+        'Example 3: \n'
+        'Question: {}?\nOptions: {}\nAnswer: {}\nYour output: '
+    )
+    return tmpl.format(question, options, prediction)
+
+
+def build_prompt_cn(question, options, prediction):
+    tmpl = (
+        '你是一个帮助我匹配答案与单选题中多个选项的 AI 助手。'
+        '你会被提供：一个问题，多个选项，一个答案。你的任务是找到与答案意义最相近的选项。'
+        '如果所有选项的意义都与答案显著不同，则输出 Z。'
+        '你应该输出一个单个的大写字母，例如 A, B, C, D（如果它们是有效选项），或 Z。'
+        '例 1:'
+        '问题: 图中最主要的物体是什么?\n选项: A. 泰迪熊 B. 兔子 C. 猫 D. 狗\n答案: 一只可爱的泰迪熊\n输出: A\n'
+        '例 2: \n'
+        '问题: 图中最主要的物体是什么?\n选项: A. 泰迪熊 B. 兔子 C. 猫 D. 狗\n答案: 蜘蛛\n输出: Z\n'
+        '例 3: \n'
+        '问题: {}?\n选项: {}\n答案: {}\n输出: '
+    )
+    return tmpl.format(question, options, prediction)
+
+
+def build_choices(item):
+    ret = {}
+    for ch in string.ascii_uppercase:
+        if ch in item and (not pd.isna(item[ch])):
+            ret[ch] = item[ch]
+    return ret
+
+
+def prefetch_answer(item):
+    choices = build_choices(item)
+    return can_infer(item['prediction'], choices)
+
+
+def extract_answer_from_item(model, item):
+    logger = get_logger('Evaluation')
+    # It will return: (pred, raw, llm_time)
+    choices = build_choices(item)
+    option_str = build_option_str(choices)
+
+    if cn_string(item['question']):
+        prompt = build_prompt_cn(item['question'], option_str, item['prediction'])
+    else:
+        prompt = build_prompt(item['question'], option_str, item['prediction'])
+    retry = 3
+
+    ret = can_infer(item['prediction'], choices)
+    if ret:
+        return dict(opt=ret, log=item['prediction'])
+
+    while retry:
+        ans = model.generate(prompt)
+        if 'Failed to obtain answer via API' in ans:
+            logger.warning('GPT API failed to answer. ')
+        else:
+            ret = can_infer(ans, choices)
+            if ret:
+                return dict(opt=ret, log=ans)
+            else:
+                logger.warning(f'Output includes 0 / > 1 letter among candidates {set(choices)} and Z: {ans}')
+        retry -= 1
+
+        if retry == 0:
+            options = list(choices) + ['Z'] if 'Z' not in choices else []
+            return dict(opt=rd.choice(options), log='Failed to predict, thus randomly generate one. ')
+
+
+def prefetch_sub_data(sub_data, answer_map, verbose=False):
+    lt = len(sub_data)
+    GT, PRED = [], []
+    for i in range(lt):
+        item = sub_data.iloc[i]
+        idx = item['index']
+        GT.append(answer_map[idx])
+        PRED.append(prefetch_answer(item))
+        if PRED[-1] and (GT[-1] != PRED[-1]):
+            log = (
+                f'Failed in Prefetching Rolling {i}: Answer is {GT[-1]}, '
+                f"Prediction is {item['prediction']}, Pre-fetched is {PRED[-1]}. "
+            )
+            return dict(hit=0, log=log)
+    flag = True
+    for g, p in zip(GT, PRED):
+        if g != p:
+            flag = False
+    ret = (dict(hit=1, log='Succeed During Pre-fetching'), ) if flag else (None, )
+    ret = ret + (GT, PRED) if verbose else ret
+    return ret if len(ret) > 1 else ret[0]
+
+
+def eval_sub_data(model, sub_data, answer_map):
+    res, GT, PRED = prefetch_sub_data(sub_data, answer_map, verbose=True)
+    if res is not None:
+        return res
+
+    lt = len(sub_data)
+    log = ''
+    for i in range(lt):
+        if PRED[i]:
+            log += f'Rolling {i} Matched.\n'
+        else:
+            res = extract_answer_from_item(model, sub_data.iloc[i])
+            opt, match_log = res['opt'], res['log']
+            PRED[i] = opt
+            if PRED[i] != GT[i]:
+                log += (
+                    f"Failed in Rolling {i}: Answer is {GT[i]}; Prediction is {sub_data.iloc[i]['prediction']}; "
+                    f'Pre-fetched is {PRED[i]}; Match Log is {match_log}.\n'
+                )
+                return dict(hit=0, log=log)
+            else:
+                log += (
+                    f"Rolling {i}: Answer is {GT[i]}, Prediction is {sub_data.iloc[i]['prediction']}, "
+                    f'Pre-fetched is {PRED[i]}.\n'
+                )
+
+    return dict(hit=1, log=log)
+
+
+def eval_data_groups(model, data_groups, answer_map, result, result_file, nproc=16):
+    prefetched = [prefetch_sub_data(g, answer_map, verbose=False) for g in data_groups]
+    remain = []
+    for dg, pf in zip(data_groups, prefetched):
+        if pf:
+            result[dg.iloc[0]['index'] % 1e6] = pf
+        else:
+            remain.append(dg)
+    dump(result, result_file)
+    tups = [(model, x, answer_map) for x in remain]
+    keys = [x.iloc[0]['index'] % 1e6 for x in remain]
+    if len(tups) == 0:
+        return
+
+    if model is None:
+        logger = get_logger('Evaluation')
+        logger.warning('Exact Matching mode, will not do GPT-based answer matching. ')
+        for k in keys:
+            result[k] = dict(
+                hit=0, log='Failed in Prefetch, no GPT-based answer matching under `exact_matching` policy.')
+        dump(result, result_file)
+        return
+
+    res = track_progress_rich(
+        eval_sub_data,
+        tups,
+        nproc=nproc,
+        chunksize=nproc,
+        save=result_file,
+        keys=keys)
+    result = load(result_file)
+    for k, v in zip(keys, res):
+        if k in result:
+            assert result[k]['hit'] == v['hit'] and result[k]['log'] == v['log']
+        else:
+            result[k] = v
+    dump(result, result_file)
+
+
+def multiple_choice_eval(eval_file, dataset='default', **judge_kwargs):
+    logger = get_logger('Evaluation')
+
+    # assert dataset is not None
+    dataset_map = {
+        'MMBench_TEST_EN': 'MMBench', 'MMBench_TEST_EN_V11': 'MMBench_V11',
+        'MMBench_TEST_CN': 'MMBench_CN', 'MMBench_TEST_CN_V11': 'MMBench_CN_V11'
+    }
+    if dataset in dataset_map:
+        dataset = dataset_map[dataset]
+    nproc = judge_kwargs.pop('nproc', 4)
+
+    if listinstr(['mmbench', 'ccbench'], dataset.lower()):
+        data = load(eval_file)
+        data['index'] = [int(x) for x in data['index']]
+        dump(data, eval_file)
+
+    rd.seed(2680)
+    suffix = eval_file.split('.')[-1]
+    model = judge_kwargs['model']
+    assert model in ['chatgpt-0613', 'exact_matching', 'gpt-4-0125']
+    name_str_map = {
+        'chatgpt-0613': 'openai',
+        'gpt-4-0125': 'gpt4'
+    }
+    name_str = name_str_map[model] if model in name_str_map else model
+
+    if model == 'exact_matching':
+        model = None
+    else:
+        if INTERNAL or gpt_key_set():
+            model = build_judge(**judge_kwargs)
+        else:
+            logger.error('OPENAI_API_KEY is not set properly, will use exact matching for evaluation')
+            model = None
+
+    logger.info(f'Evaluating {eval_file}')
+    result_file = eval_file.replace(f'.{suffix}', f'_{name_str}_result.pkl')
+    result = {}
+    if osp.exists(result_file):
+        result = load(result_file)
+
+    data = load(eval_file)
+    data = data.sort_values(by='index')
+    data['prediction'] = [str(x) for x in data['prediction']]
+    for k in data.keys():
+        data[k.lower() if k not in list(string.ascii_uppercase) else k] = data.pop(k)
+
+    if dataset != 'default':
+        meta = TSVDataset(dataset).data
+    else:
+        logger.warning('Dataset is not provided, try to use the original `eval_file` as meta data. ')
+        meta = load(eval_file)
+        assert 'index' in meta and 'answer' in meta, 'Essentail columns missing in the eval_file.'
+
+    answer_map = {i: c for i, c in zip(meta['index'], meta['answer'])}
+    cate_map = {i: c for i, c in zip(meta['index'], meta['category'])} if 'category' in meta else None
+    l2_cate_map = {i: c for i, c in zip(meta['index'], meta['l2-category'])} if 'l2-category' in meta else None
+    split_map = {i: c for i, c in zip(meta['index'], meta['split'])} if 'split' in meta else None
+
+    if cate_map is not None and np.all([pd.isna(x) for x in cate_map.values()]):
+        cate_map = None
+    if l2_cate_map is not None and np.all([pd.isna(x) for x in l2_cate_map.values()]):
+        l2_cate_map = None
+    if split_map is not None and np.all([pd.isna(x) for x in split_map.values()]):
+        split_map = None
+
+    if listinstr(['MMMU'], dataset):
+        data = MMMU_preproc(data)
+        answer_map = {k: (v if v in list(string.ascii_uppercase) else 'A') for k, v in answer_map.items()}
+
+    data = data[data['index'].isin(answer_map)]
+    data_main = data[data['index'] < int(1e6)]
+    meta_idx_set = set(meta['index'])
+    data_main = data_main[data_main['index'].isin(meta_idx_set)]
+
+    lt = len(data_main)
+    hit, tot = 0, 0
+
+    data_groups = []
+    for i in tqdm(range(lt)):
+        # Dealing with the normal part
+        item_main = data_main.iloc[i]
+        idx = item_main['index']
+
+        if idx in result:
+            correct = result[idx]['hit']
+            assert correct in [0, 1]
+            hit += correct
+            tot += 1
+            continue
+
+        sub_data = data[data['index'] % int(1e6) == idx]
+        data_groups.append(sub_data)
+
+    if len(data_groups):
+        eval_data_groups(
+            model=model,
+            data_groups=data_groups,
+            answer_map=answer_map,
+            nproc=nproc,
+            result=result,
+            result_file=result_file)
+
+    tmp_pth = f'/tmp/{timestr()}.xlsx'
+    dump(data_main, tmp_pth)
+    data_main = load(tmp_pth)
+
+    res = load(result_file)
+    indices = data_main['index']
+
+    data_main['hit'] = [res[i]['hit'] for i in indices]
+    data_main['log'] = [res[i]['log'] for i in indices]
+
+    main_idx = data_main['index']
+    if cate_map is not None:
+        data_main['category'] = [cate_map[i] for i in main_idx]
+    if l2_cate_map is not None:
+        data_main['l2-category'] = [l2_cate_map[i] for i in main_idx]
+    if split_map is not None:
+        data_main['split'] = [split_map[i] for i in indices]
+
+    # load split
+    dump(data_main, eval_file.replace(f'.{suffix}', f'_{name_str}_result.{suffix}'))
+    data_main = load(eval_file.replace(f'.{suffix}', f'_{name_str}_result.{suffix}'))
+
+    acc = report_acc(data_main)
+    score_file = eval_file.replace(f'.{suffix}', '_acc.csv')
+    dump(acc, score_file)
+    logger.info(f'multiple_choice_eval successfully finished evaluating {eval_file}, results saved in {score_file}')
+    logger.info('Score: ')
+    logger.info(acc)
+    return acc
+
+
+def parse_args():
+    parser = argparse.ArgumentParser(description='Inference LLM Answers. ')
+    parser.add_argument('data', type=str, help='The question set for inference, in excel / tsv / json format. ')
+    parser.add_argument(
+        '--model',
+        type=str,
+        help='The LLM (GPT) used for inference. ',
+        default='chatgpt-0613',
+        choices=['chatgpt-0613', 'exact_matching', 'gpt-4-0125'])
+    parser.add_argument(
+        '--dataset',
+        type=str,
+        default='default',
+        help='The dataset to evaluate')
+    parser.add_argument('--nproc', type=int, default=6)
+    parser.add_argument('--verbose', action='store_true')
+    args = parser.parse_args()
+    return args
+
+
+if __name__ == '__main__':
+    load_env()
+    args = parse_args()
+    judge_kwargs = dict(model=args.model, nproc=args.nproc, verbose=args.verbose)
+    if 'OPENAI_API_KEY_JUDGE' in os.environ and os.environ['OPENAI_API_KEY_JUDGE']:
+        judge_kwargs['key'] = os.environ['OPENAI_API_KEY_JUDGE']
+    if 'OPENAI_API_BASE_JUDGE' in os.environ and os.environ['OPENAI_API_BASE_JUDGE']:
+        judge_kwargs['api_base'] = os.environ['OPENAI_API_BASE_JUDGE']
+    acc = multiple_choice_eval(eval_file=args.data, dataset=args.dataset, **judge_kwargs)
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/vqa_eval.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/vqa_eval.py
new file mode 100644
index 0000000000000000000000000000000000000000..33dc01d9662692cbf1c9b726feccdfd0f9ac9f84
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/vqa_eval.py
@@ -0,0 +1,340 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+# Partly adopted from https://github.com/GT-Vision-Lab/VQA
+# Copyright (c) 2014, Aishwarya Agrawal
+
+import re
+from vlmeval.smp import *
+from typing import Optional
+from functools import partial
+
+
+def _process_digit_article(inText):
+    outText = []
+    tempText = inText.lower().split()
+    articles = ['a', 'an', 'the']
+    manualMap = {
+        'none': '0',
+        'zero': '0',
+        'one': '1',
+        'two': '2',
+        'three': '3',
+        'four': '4',
+        'five': '5',
+        'six': '6',
+        'seven': '7',
+        'eight': '8',
+        'nine': '9',
+        'ten': '10',
+    }
+    contractions = {
+        'aint': "ain't",
+        'arent': "aren't",
+        'cant': "can't",
+        'couldve': "could've",
+        'couldnt': "couldn't",
+        "couldn'tve": "couldn't've",
+        "couldnt've": "couldn't've",
+        'didnt': "didn't",
+        'doesnt': "doesn't",
+        'dont': "don't",
+        'hadnt': "hadn't",
+        "hadnt've": "hadn't've",
+        "hadn'tve": "hadn't've",
+        'hasnt': "hasn't",
+        'havent': "haven't",
+        'hed': "he'd",
+        "hed've": "he'd've",
+        "he'dve": "he'd've",
+        'hes': "he's",
+        'howd': "how'd",
+        'howll': "how'll",
+        'hows': "how's",
+        "Id've": "I'd've",
+        "I'dve": "I'd've",
+        'Im': "I'm",
+        'Ive': "I've",
+        'isnt': "isn't",
+        'itd': "it'd",
+        "itd've": "it'd've",
+        "it'dve": "it'd've",
+        'itll': "it'll",
+        "let's": "let's",
+        'maam': "ma'am",
+        'mightnt': "mightn't",
+        "mightnt've": "mightn't've",
+        "mightn'tve": "mightn't've",
+        'mightve': "might've",
+        'mustnt': "mustn't",
+        'mustve': "must've",
+        'neednt': "needn't",
+        'notve': "not've",
+        'oclock': "o'clock",
+        'oughtnt': "oughtn't",
+        "ow's'at": "'ow's'at",
+        "'ows'at": "'ow's'at",
+        "'ow'sat": "'ow's'at",
+        'shant': "shan't",
+        "shed've": "she'd've",
+        "she'dve": "she'd've",
+        "she's": "she's",
+        'shouldve': "should've",
+        'shouldnt': "shouldn't",
+        "shouldnt've": "shouldn't've",
+        "shouldn'tve": "shouldn't've",
+        "somebody'd": 'somebodyd',
+        "somebodyd've": "somebody'd've",
+        "somebody'dve": "somebody'd've",
+        'somebodyll': "somebody'll",
+        'somebodys': "somebody's",
+        'someoned': "someone'd",
+        "someoned've": "someone'd've",
+        "someone'dve": "someone'd've",
+        'someonell': "someone'll",
+        'someones': "someone's",
+        'somethingd': "something'd",
+        "somethingd've": "something'd've",
+        "something'dve": "something'd've",
+        'somethingll': "something'll",
+        'thats': "that's",
+        'thered': "there'd",
+        "thered've": "there'd've",
+        "there'dve": "there'd've",
+        'therere': "there're",
+        'theres': "there's",
+        'theyd': "they'd",
+        "theyd've": "they'd've",
+        "they'dve": "they'd've",
+        'theyll': "they'll",
+        'theyre': "they're",
+        'theyve': "they've",
+        'twas': "'twas",
+        'wasnt': "wasn't",
+        "wed've": "we'd've",
+        "we'dve": "we'd've",
+        'weve': "we've",
+        'werent': "weren't",
+        'whatll': "what'll",
+        'whatre': "what're",
+        'whats': "what's",
+        'whatve': "what've",
+        'whens': "when's",
+        'whered': "where'd",
+        'wheres': "where's",
+        'whereve': "where've",
+        'whod': "who'd",
+        "whod've": "who'd've",
+        "who'dve": "who'd've",
+        'wholl': "who'll",
+        'whos': "who's",
+        'whove': "who've",
+        'whyll': "why'll",
+        'whyre': "why're",
+        'whys': "why's",
+        'wont': "won't",
+        'wouldve': "would've",
+        'wouldnt': "wouldn't",
+        "wouldnt've": "wouldn't've",
+        "wouldn'tve": "wouldn't've",
+        'yall': "y'all",
+        "yall'll": "y'all'll",
+        "y'allll": "y'all'll",
+        "yall'd've": "y'all'd've",
+        "y'alld've": "y'all'd've",
+        "y'all'dve": "y'all'd've",
+        'youd': "you'd",
+        "youd've": "you'd've",
+        "you'dve": "you'd've",
+        'youll': "you'll",
+        'youre': "you're",
+        'youve': "you've",
+    }
+    for word in tempText:
+        word = manualMap.setdefault(word, word)
+        if word not in articles:
+            outText.append(word)
+    for wordId, word in enumerate(outText):
+        if word in contractions:
+            outText[wordId] = contractions[word]
+    outText = ' '.join(outText)
+    return outText
+
+
+def hit_calculate(result, dataset_name, anls_threshold=0.5):
+    if listinstr(['TextVQA'], dataset_name):
+        return [np.mean(x['match']) for x in result]
+    elif listinstr(['DocVQA', 'InfoVQA'], dataset_name):
+        # return [1 - np.min(x['match']) >= anls_threshold for x in result]
+        return [0.0 if 1 - np.min(x['match']) < anls_threshold else 1 - np.min(x['match']) for x in result]
+    elif listinstr(['ChartQA', 'OCRVQA'], dataset_name):
+        return [np.max(x['match']) for x in result]
+    else:  # default using vqa_score to calculate score
+        return [np.mean(x['match']) for x in result]
+
+
+# https://github.com/google-research/pix2struct/blob/main/pix2struct/metrics.py#L81
+def relaxed_correctness(target: str,
+                        prediction: str,
+                        max_relative_change: float = 0.05) -> bool:
+    """Calculates relaxed correctness.
+
+    The correctness tolerates certain error ratio defined by max_relative_change.
+    See https://arxiv.org/pdf/2203.10244.pdf, end of section 5.1:
+    “Following Methani et al. (2020), we use a relaxed accuracy measure for the
+    numeric answers to allow a minor inaccuracy that may result from the automatic
+    data extraction process. We consider an answer to be correct if it is within
+    5% of the gold answer. For non-numeric answers, we still need an exact match
+    to consider an answer to be correct.”
+
+    Args:
+      target: Target string.
+      prediction: Predicted string.
+      max_relative_change: Maximum relative change.
+
+    Returns:
+      Whether the prediction was correct given the specified tolerance.
+    """
+
+    def _to_float(text: str) -> Optional[float]:
+        try:
+            if text.endswith('%'):
+                # Convert percentages to floats.
+                return float(text.rstrip('%')) / 100.0
+            else:
+                return float(text)
+        except ValueError:
+            return None
+    prediction = str(prediction)
+    target = str(target)
+    prediction_float = _to_float(prediction)
+    target_float = _to_float(target)
+    if prediction_float is not None and target_float:
+        relative_change = abs(prediction_float - target_float) / abs(target_float)
+        return relative_change <= max_relative_change
+    else:
+        return prediction.lower() == target.lower()
+
+
+def levenshtein_distance(s1, s2):
+    if len(s1) > len(s2):
+        s1, s2 = s2, s1
+
+    distances = range(len(s1) + 1)
+    for i2, c2 in enumerate(s2):
+        distances_ = [i2 + 1]
+        for i1, c1 in enumerate(s1):
+            if c1 == c2:
+                distances_.append(distances[i1])
+            else:
+                distances_.append(1 + min((distances[i1], distances[i1 + 1], distances_[-1])))
+        distances = distances_
+    return distances[-1]
+
+
+def anls_compute(groundtruth, prediction):
+    gt_answer = ' '.join(groundtruth.strip().lower().split())
+    det_answer = ' '.join(prediction.strip().lower().split())
+    dist = levenshtein_distance(gt_answer, det_answer)
+    length = max(len(groundtruth.upper()), len(prediction.upper()))
+    values = 0.0 if length == 0 else float(dist) / float(length)
+    return values
+
+
+def process_answer(answer):
+    answer = answer.replace('\n', ' ')
+    answer = answer.replace('\t', ' ')
+    answer = answer.strip()
+    answer = process_punctuation(answer)
+    answer = _process_digit_article(answer)
+    return answer
+
+
+def process_line(line, method='vqa_score'):
+    ret = {}
+    if istype(line['answer'], list):
+        answers = eval(line['answer'])
+    else:
+        answers = [line['answer']]
+    if method == 'vqa_score':
+        ret['gt'] = [process_answer(x) for x in answers]
+        ret['pred'] = process_answer(line['prediction'])
+        ret['match'] = []
+        for current_idx, gtAnsDatum in enumerate(ret['gt']):
+            otherGTAns = [
+                item for ret_gt_idx, item in enumerate(ret['gt'])
+                if ret_gt_idx != current_idx
+            ]
+            matchingAns = [
+                item for item in otherGTAns if item == ret['pred']
+            ]
+            acc = min(1, float(len(matchingAns)) / 3)
+            ret['match'].append(acc)
+    elif method == 'anls':
+        ret['gt'] = answers
+        ret['pred'] = line['prediction']
+        ret['match'] = [anls_compute(x, ret['pred']) for x in ret['gt']]
+    elif method == 'relaxed_accuracy':
+        ret['gt'] = answers
+        ret['pred'] = line['prediction'].strip()
+        ret['match'] = [relaxed_correctness(ret['pred'], x) for x in ret['gt']]
+    elif method == 'accuracy':
+        ret['gt'] = answers
+        ret['pred'] = line['prediction'].strip()
+        ret['match'] = [(1.0 if (x.strip().lower() == ret['pred'].strip().lower()) else 0.0) for x in ret['gt']]
+    else:  # default using vqa_score to calculate score
+        ret['gt'] = [process_answer(x) for x in answers]
+        ret['pred'] = process_answer(line['prediction'])
+        ret['match'] = [x == ret['pred'] for x in ret['gt']]
+
+    return ret
+
+
+def VQAEval(eval_file, dataset_name, **kwargs):
+    logger = get_logger('Evaluation')
+    data = load(eval_file)
+    assert 'answer' in data and 'prediction' in data
+    data['prediction'] = [str(x) for x in data['prediction']]
+    data['answer'] = [str(x) for x in data['answer']]
+    lt = len(data)
+    pool = mp.Pool(16)
+    lines = [data.iloc[i] for i in range(lt)]
+    if listinstr(['TextVQA'], dataset_name):
+        res = pool.map(partial(process_line, method='vqa_score'), lines)
+    elif listinstr(['ChartQA'], dataset_name):
+        res = pool.map(partial(process_line, method='relaxed_accuracy'), lines)
+    elif listinstr(['OCRVQA'], dataset_name):
+        res = pool.map(partial(process_line, method='accuracy'), lines)
+    elif listinstr(['DocVQA', 'InfoVQA'], dataset_name):
+        res = pool.map(partial(process_line, method='anls'), lines)
+    else:  # default using vqa_score to calculate score
+        res = pool.map(process_line, lines)
+    # [np.mean(x['match']) >= full_score_weight for x in res]
+    hit = hit_calculate(res, dataset_name)
+    ret = dict()
+    if 'split' in data:
+        splits = set(data['split'])
+        for sp in splits:
+            sub = [r for l, r in zip(lines, res) if l['split'] == sp]
+            # [np.mean(x['match']) >= full_score_weight for x in sub]
+            hit = hit_calculate(sub, dataset_name)
+            ret[sp] = np.mean(hit) * 100
+        sub = [r for l, r in zip(lines, res)]
+        hit = hit_calculate(sub, dataset_name)
+        ret['Overall'] = np.mean(hit) * 100
+    else:
+        ret['Overall'] = np.mean(hit) * 100
+        if 'category' in data:
+            cates = list(set(data['category']))
+            cates.sort()
+            for c in cates:
+                sub = [r for l, r in zip(lines, res) if l['category'] == c]
+                # [np.mean(x['match']) >= full_score_weight for x in sub]
+                hit = hit_calculate(sub, dataset_name)
+                ret[c] = np.mean(hit) * 100
+    ret = d2df(ret)
+    ret.round(2)
+
+    suffix = eval_file.split('.')[-1]
+    result_file = eval_file.replace(f'.{suffix}', '_acc.csv')
+    logger.info(f'VQA Eval Finished. Saved to {result_file}. ')
+    logger.info(ret)
+    dump(ret, result_file)
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/yes_or_no.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/yes_or_no.py
new file mode 100644
index 0000000000000000000000000000000000000000..6012c3745842c2b4d29674e05b13afb0aa9ff4b0
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/yes_or_no.py
@@ -0,0 +1,297 @@
+from vlmeval.evaluate.misc import build_judge
+from vlmeval.smp import *
+from vlmeval.utils import track_progress_rich
+
+INTERNAL = os.environ.get('INTERNAL', 0)
+
+
+def MME_rating(data_file):
+    data = load(data_file)
+    stats = defaultdict(dict)
+    lt = len(data)
+    for i in range(lt):
+        item = data.iloc[i]
+        category = item['category']
+        image_path = item['image_path']
+        score = item['score']
+        if image_path not in stats[category]:
+            stats[category][image_path] = []
+        stats[category][image_path].append(score)
+
+    def acc(key, mode='normal'):
+        res = stats[key]
+        values = []
+        for val in res.values():
+            if mode == 'normal':
+                values.extend(val)
+            elif mode == 'plus':
+                values.append(val[0] * val[1])
+        return np.mean(values) * 100
+
+    scores = {}
+    for k in stats:
+        scores[k] = acc(k) + acc(k, 'plus')
+
+    super_cates = dict(
+        perception=[
+            'OCR', 'artwork', 'celebrity', 'color', 'count', 'existence',
+            'landmark', 'position', 'posters', 'scene'
+        ],
+        reasoning=['code_reasoning', 'commonsense_reasoning', 'numerical_calculation', 'text_translation']
+    )
+
+    ret = {}
+    for sc, cate_list in super_cates.items():
+        base = 0
+        for c in cate_list:
+            base += scores[c]
+        ret[sc] = base
+    ret.update(scores)
+    ret = d2df(ret)
+    return ret
+
+
+def Hallusion_rating(data_file):
+    def calc_fAcc(data):
+        res = defaultdict(list)
+        lt = len(data)
+        for i in range(lt):
+            line = data.iloc[i]
+            res[f"{line['l2-category']}_{line['set_id']}_{line['figure_id']}"].append(line['score'])
+        return np.mean([np.all(x) for x in res.values()]) * 100
+
+    def calc_qAcc(data):
+        res = defaultdict(list)
+        lt = len(data)
+        for i in range(lt):
+            line = data.iloc[i]
+            res[f"{line['l2-category']}_{line['set_id']}_{line['question_id']}"].append(line['score'])
+        return np.mean([np.all(x) for x in res.values()]) * 100
+
+    def calc_aAcc(data):
+        return np.mean(data['score']) * 100
+
+    data = load(data_file)
+    data['set_id'] = [x.split('_')[3] for x in data['index']]
+    data['figure_id'] = [x.split('_')[4] for x in data['index']]
+    data['question_id'] = [x.split('_')[5] for x in data['index']]
+
+    res = dict(split=[], aAcc=[], fAcc=[], qAcc=[])
+    res['split'].append('Overall')
+    res['aAcc'].append(calc_aAcc(data))
+    res['fAcc'].append(calc_fAcc(data))
+    res['qAcc'].append(calc_qAcc(data))
+
+    if 'category' in data:
+        cates = list(set(data['category']))
+        for c in cates:
+            sub = data[data['category'] == c]
+            res['split'].append(c)
+            res['aAcc'].append(calc_aAcc(sub))
+            res['fAcc'].append(calc_fAcc(sub))
+            res['qAcc'].append(calc_qAcc(sub))
+
+    if 'l2-category' in data:
+        cates = list(set(data['l2-category']))
+        for c in cates:
+            sub = data[data['l2-category'] == c]
+            res['split'].append(c)
+            res['aAcc'].append(calc_aAcc(sub))
+            res['fAcc'].append(calc_fAcc(sub))
+            res['qAcc'].append(calc_qAcc(sub))
+    ret = pd.DataFrame(res)
+    return ret
+
+
+def POPE_rating(data_file):
+    def cal_f1_score(y_true, y_pred):
+        tp = sum((y_true == 1) & (y_pred == 1))
+        fp = sum((y_true == 0) & (y_pred == 1))
+        fn = sum((y_true == 1) & (y_pred == 0))
+
+        precision = tp / (tp + fp) if (tp + fp) != 0 else 0
+        recall = tp / (tp + fn) if (tp + fn) != 0 else 0
+        f1_score = 2 * (precision * recall) / (precision + recall) if (precision + recall) != 0 else 0
+        return f1_score, precision, recall
+
+    data = load(data_file)
+    data = data.assign(category=data['category'].str.split(',')).explode('category')
+    data['index'] = range(len(data))
+    res = dict(split=[], Overall=[], acc=[], precision=[], recall=[])
+    y_true = np.array([1 if i == 'Yes' else 0 for i in data['answer']])
+    y_pred = np.array([1 if i == 'Yes' else 0 for i in data['extracted']])
+    f1_score, precision, recall = cal_f1_score(y_true, y_pred)
+    res['split'].append('Overall')
+    res['Overall'].append(f1_score * 100)
+    res['acc'].append(np.mean(data['score']) * 100)
+    res['precision'].append(precision * 100)
+    res['recall'].append(recall * 100)
+
+    if 'category' in data:
+        cates = list(set(data['category']))
+        cates = [c for c in cates if not pd.isna(c)]
+        for c in cates:
+            sub = data[data['category'] == c]
+            y_true = np.array([1 if i == 'Yes' else 0 for i in sub['answer']])
+            y_pred = np.array([1 if i == 'Yes' else 0 for i in sub['extracted']])
+            f1_score, precision, recall = cal_f1_score(y_true, y_pred)
+            res['split'].append(c)
+            res['Overall'].append(f1_score * 100)
+            res['acc'].append(np.mean(sub['score']) * 100)
+            res['precision'].append(precision * 100)
+            res['recall'].append(recall * 100)
+
+    ret = pd.DataFrame(res)
+    return ret
+
+
+def default_rating(data_file):
+    data = load(data_file)
+    res = {}
+    res['Overall'] = np.mean(data['score']) * 100
+    if 'category' in data:
+        cates = list(set(data['category']))
+        cates = [c for c in cates if not pd.isna(c)]
+        cates.sort()
+        for c in cates:
+            sub = data[data['category'] == c]
+            res[c] = np.mean(sub['score']) * 100
+    if 'l2-category' in data:
+        cates = list(set(data['l2-category']))
+        cates = [c for c in cates if not pd.isna(c)]
+        cates.sort()
+        for c in cates:
+            sub = data[data['l2-category'] == c]
+            res[c] = np.mean(sub['score']) * 100
+    ret = d2df(res)
+    return ret
+
+
+def YOrN_match_prompt(line):
+    tmpl = (
+        'You are an AI assistant who will help me to match an answer with two options of a question. '
+        'The options are only Yes / No. '
+        'You are provided with a question and an answer, '
+        'and you need to find which option (Yes / No) is most similar to the answer. '
+        'If the meaning of all options are significantly different from the answer, output Unknown. '
+        'Your should output a single word among the following 3 choices: Yes, No, Unknown.\n'
+        'Example 1: \n'
+        "Question: Is the word in this image 'Hello'?\nAnswer: The word in this image is 'Hello'.\nYour output: Yes\n"
+        'Example 2: \n'
+        "Question: Is the word in this image 'Hello'?\n"
+        "Answer: The word in this image is not 'Hello'.\nYour output: No\n"
+        'Example 3: \n'
+        'Question: {}?\nAnswer: {}\nYour output: '
+    )
+    return tmpl.format(line['question'], line['prediction'])
+
+
+def YOrN_Extraction(output):
+    s = output.lower()
+    words = process_punctuation(s).split()
+    if 'yes' in words and 'no' not in words:
+        return 'Yes'
+    if 'yes' not in words and 'no' in words:
+        return 'No'
+    return 'Unknown'
+
+
+def YOrN_auxeval(model, line):
+    prompt = YOrN_match_prompt(line)
+    retry = 5
+    for i in range(retry):
+        output = model.generate(prompt, temperature=0.5 * i)
+        ans = YOrN_Extraction(output)
+        if ans != 'Unknown':
+            return ans
+    return 'Unknown'
+
+
+def YOrN_eval(eval_file, dataset=None, **judge_kwargs):
+    logger = get_logger('Evaluation')
+    data = load(eval_file)
+    data['prediction'] = [str(x) for x in data['prediction']]
+    storage = eval_file.replace('.xlsx', '_auxmatch.xlsx')
+    tmp_file = eval_file.replace('.xlsx', '_tmp.pkl')
+    nproc = judge_kwargs.pop('nproc', 4)
+
+    if not osp.exists(storage):
+        ans_map = {k: YOrN_Extraction(v) for k, v in zip(data['index'], data['prediction'])}
+        if osp.exists(tmp_file):
+            tmp = load(tmp_file)
+            for k in tmp:
+                if ans_map[k] == 'Unknown' and tmp[k] != 'Unknown':
+                    ans_map[k] = tmp[k]
+
+        data['extracted'] = [ans_map[x] for x in data['index']]
+        unknown = data[data['extracted'] == 'Unknown']
+
+        if INTERNAL or gpt_key_set():
+            model = build_judge(**judge_kwargs)
+        else:
+            logger.error('OPENAI_API_KEY is not set properly, will use exact matching for evaluation')
+            model = None
+
+        if model is not None:
+            lt = len(unknown)
+            lines = [unknown.iloc[i] for i in range(lt)]
+            tups = [(model, line) for line in lines]
+            indices = list(unknown['index'])
+            if len(tups):
+                res = track_progress_rich(
+                    YOrN_auxeval, tups, nproc=nproc, chunksize=nproc, keys=indices, save=tmp_file)
+                for k, v in zip(indices, res):
+                    ans_map[k] = v
+
+        data['extracted'] = [ans_map[x] for x in data['index']]
+        dump(data, storage)
+    else:
+        logger.warning(f'GPT matching file {storage} already exists, will reuse it in YOrN_eval. ')
+
+    data = load(storage)
+    data['score'] = (data['answer'] == data['extracted'])
+    dump(data, storage)
+
+    if dataset is not None and listinstr(['MME'], dataset):
+        score = MME_rating(storage)
+    elif dataset is not None and listinstr(['Hallusion'], dataset):
+        score = Hallusion_rating(storage)
+    elif dataset is not None and listinstr(['POPE'], dataset):
+        score = POPE_rating(storage)
+    else:
+        score = default_rating(storage)
+
+    score_tgt = eval_file.replace('.xlsx', '_score.csv')
+    dump(score, score_tgt)
+
+    logger.info(f'YOrN_eval successfully finished evaluating {eval_file}, results saved in {score_tgt}')
+    logger.info('Score: ')
+    logger.info(score)
+    return score
+
+
+def parse_args():
+    parser = argparse.ArgumentParser(description='Inference LLM Answers. ')
+    parser.add_argument('data', type=str, help='The question set for inference, in excel / tsv / json format. ')
+    parser.add_argument(
+        '--model',
+        type=str,
+        help='The LLM (GPT) used for inference. ',
+        default='chatgpt-0613',
+        choices=['chatgpt-0613'])
+    parser.add_argument('--nproc', type=int, default=4)
+    parser.add_argument('--dataset', type=str, default=None)
+    parser.add_argument('--verbose', action='store_true')
+    args = parser.parse_args()
+    return args
+
+
+if __name__ == '__main__':
+    load_env()
+    args = parse_args()
+    judge_kwargs = dict(model=args.model, nproc=args.nproc, verbose=args.verbose)
+    if 'OPENAI_API_KEY_JUDGE' in os.environ and os.environ['OPENAI_API_KEY_JUDGE']:
+        judge_kwargs['key'] = os.environ['OPENAI_API_KEY_JUDGE']
+    if 'OPENAI_API_BASE_JUDGE' in os.environ and os.environ['OPENAI_API_BASE_JUDGE']:
+        judge_kwargs['api_base'] = os.environ['OPENAI_API_BASE_JUDGE']
+    acc = YOrN_eval(eval_file=args.data, dataset=args.dataset, **judge_kwargs)
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/inference.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/inference.py
new file mode 100644
index 0000000000000000000000000000000000000000..4719c2e42346c58ff449bd8f8ee3ba48590b78e2
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/inference.py
@@ -0,0 +1,184 @@
+import torch
+import torch.distributed as dist
+import datetime
+from vlmeval.config import supported_VLM, api_models
+from vlmeval.utils import TSVDataset, track_progress_rich, split_MMMU
+from vlmeval.smp import *
+
+FAIL_MSG = 'Failed to obtain answer via API.'
+
+
+def parse_args():
+    parser = argparse.ArgumentParser()
+    parser.add_argument('--data', type=str, nargs='+', required=True)
+    parser.add_argument('--model', type=str, nargs='+', required=True)
+    parser.add_argument('--nproc', type=int, default=4, required=True)
+    parser.add_argument('--verbose', action='store_true')
+    args = parser.parse_args()
+    return args
+
+
+# Only API model is accepted
+def infer_data_api(work_dir, model_name, dataset_name, index_set=None, api_nproc=4, ignore_failed=False):
+    rank, world_size = get_rank_and_world_size()
+    assert rank == 0 and world_size == 1
+    dataset = TSVDataset(dataset_name)
+    data = dataset.data
+    if index_set is not None:
+        data = data[data['index'].isin(index_set)]
+
+    model = supported_VLM[model_name]() if isinstance(model_name, str) else model_name
+    assert getattr(model, 'is_api', False)
+
+    lt, indices = len(data), list(data['index'])
+    structs = [dataset.build_prompt(data.iloc[i]) for i in range(lt)]
+
+    # Corner Case
+    if listinstr(['MMMU'], dataset_name):
+        structs = [split_MMMU(s) for s in structs]
+
+    out_file = f'{work_dir}/{model_name}_{dataset_name}_supp.pkl'
+    res = {}
+    if osp.exists(out_file):
+        res = load(out_file)
+        if ignore_failed:
+            res = {k: v for k, v in res.items() if FAIL_MSG not in v}
+
+    structs = [s for i, s in zip(indices, structs) if i not in res]
+    indices = [i for i in indices if i not in res]
+
+    gen_func = model.generate
+    # For now, we do not use split_MMMU for MMMU dataset
+    structs = [dict(message=struct, dataset=dataset_name) for struct in structs]
+
+    if len(structs):
+        track_progress_rich(gen_func, structs, nproc=api_nproc, chunksize=api_nproc, save=out_file, keys=indices)
+
+    res = load(out_file)
+    if index_set is not None:
+        res = {k: v for k, v in res.items() if k in index_set}
+    os.remove(out_file)
+    return res
+
+
+def infer_data(model_name, work_dir, dataset_name, out_file, verbose=False, api_nproc=4):
+    prev_file = f'{work_dir}/{model_name}_{dataset_name}_PREV.pkl'
+    res = load(prev_file) if osp.exists(prev_file) else {}
+    if osp.exists(out_file):
+        res.update(load(out_file))
+
+    rank, world_size = get_rank_and_world_size()
+    if rank == 0:
+        dataset = TSVDataset(dataset_name)
+    if world_size > 1:
+        dist.barrier()
+    dataset = TSVDataset(dataset_name)
+
+    sheet_indices = list(range(rank, len(dataset), world_size))
+    lt = len(sheet_indices)
+    data = dataset.data.iloc[sheet_indices]
+    data_indices = [i for i in data['index']]
+
+    # If finished, will exit without building the model
+    all_finished = True
+    for i in range(lt):
+        idx = data.iloc[i]['index']
+        if idx not in res:
+            all_finished = False
+    if all_finished:
+        res = {k: res[k] for k in data_indices}
+        dump(res, out_file)
+        return
+
+    # Data need to be inferred
+    data = data[~data['index'].isin(res)]
+    lt = len(data)
+
+    model = supported_VLM[model_name]() if isinstance(model_name, str) else model_name
+
+    is_api = getattr(model, 'is_api', False)
+    if is_api:
+        lt, indices = len(data), list(data['index'])
+        supp = infer_data_api(
+            work_dir=work_dir,
+            model_name=model_name,
+            dataset_name=dataset_name,
+            index_set=set(indices),
+            api_nproc=api_nproc)
+        for idx in indices:
+            assert idx in supp
+        res.update(supp)
+        res = {k: res[k] for k in data_indices}
+        dump(res, out_file)
+        return model_name
+
+    for i in tqdm(range(lt)):
+        idx = data.iloc[i]['index']
+        if idx in res:
+            continue
+
+        if hasattr(model, 'use_custom_prompt') and model.use_custom_prompt(dataset_name):
+            struct = model.build_prompt(data.iloc[i], dataset=dataset_name)
+        else:
+            struct = dataset.build_prompt(data.iloc[i])
+
+        # Corner Case
+        if listinstr(['MMMU'], dataset_name):
+            struct = split_MMMU(struct)
+
+        # For now, we do not use split_MMMU for MMMU dataset
+        response = model.generate(message=struct, dataset=dataset_name)
+        # torch.cuda.empty_cache()
+
+        if verbose:
+            print(response, flush=True)
+
+        res[idx] = response
+        if (i + 1) % 20 == 0:
+            dump(res, out_file)
+
+    res = {k: res[k] for k in data_indices}
+    dump(res, out_file)
+    return model
+
+
+# A wrapper for infer_data, do the pre & post processing
+def infer_data_job(model, work_dir, model_name, dataset_name, verbose=False, api_nproc=4, ignore_failed=False):
+    rank, world_size = get_rank_and_world_size()
+    result_file = osp.join(work_dir, f'{model_name}_{dataset_name}.xlsx')
+
+    prev_file = f'{work_dir}/{model_name}_{dataset_name}_PREV.pkl'
+    if osp.exists(result_file):
+        if rank == 0:
+            data = load(result_file)
+            results = {k: v for k, v in zip(data['index'], data['prediction'])}
+            if not ignore_failed:
+                results = {k: v for k, v in results.items() if FAIL_MSG not in str(v)}
+            dump(results, prev_file)
+        if world_size > 1:
+            dist.barrier()
+
+    tmpl = osp.join(work_dir, '{}' + f'{world_size}_{dataset_name}.pkl')
+    out_file = tmpl.format(rank)
+
+    model = infer_data(
+        model, work_dir=work_dir, dataset_name=dataset_name, out_file=out_file, verbose=verbose, api_nproc=api_nproc)
+    if world_size > 1:
+        dist.barrier()
+
+    if rank == 0:
+        data_all = {}
+        for i in range(world_size):
+            data_all.update(load(tmpl.format(i)))
+
+        data = TSVDataset(dataset_name).data
+        for x in data['index']:
+            assert x in data_all
+        data['prediction'] = [str(data_all[x]) for x in data['index']]
+        if 'image' in data:
+            data.pop('image')
+
+        dump(data, result_file)
+        for i in range(world_size):
+            os.remove(tmpl.format(i))
+    return model
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/smp/__init__.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/smp/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..46e89687d469b83ec7dd7e3205841d35087108c7
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/smp/__init__.py
@@ -0,0 +1,4 @@
+from .file import *
+from .vlm import *
+from .misc import *
+from .log import *
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/smp/file.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/smp/file.py
new file mode 100644
index 0000000000000000000000000000000000000000..c9aa693fc8f332981ceff44dd392a7d2ded2c9de
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/smp/file.py
@@ -0,0 +1,230 @@
+import json
+import pickle
+import pandas as pd
+import os
+import csv
+import hashlib
+import os.path as osp
+import time
+import numpy as np
+import validators
+import mimetypes
+
+
+def LMUDataRoot():
+    if 'LMUData' in os.environ and osp.exists(os.environ['LMUData']):
+        return os.environ['LMUData']
+    home = osp.expanduser('~')
+    root = osp.join(home, 'LMUData')
+    os.makedirs(root, exist_ok=True)
+    # root = './LMUData'
+    # os.makedirs(root, exist_ok=True)
+    return root
+
+def MMBenchOfficialServer(dataset_name):
+    root = LMUDataRoot()
+
+    if dataset_name in ['MMBench', 'MMBench_V11', 'MMBench_CN', 'MMBench_CN_V11']:
+        ans_file = f'{root}/{dataset_name}.tsv'
+        if osp.exists(ans_file):
+            data = load(ans_file)
+            if 'answer' in data and sum([pd.isna(x) for x in data['answer']]) == 0:
+                return True
+
+    if dataset_name in ['MMBench_TEST_EN', 'MMBench_TEST_CN', 'MMBench_TEST_EN_V11', 'MMBench_TEST_CN_V11']:
+        ans_file1 = f'{root}/{dataset_name}.tsv'
+        mapp = {
+            'MMBench_TEST_EN': 'MMBench', 'MMBench_TEST_CN': 'MMBench_CN',
+            'MMBench_TEST_EN_V11': 'MMBench_V11', 'MMBench_TEST_CN_V11': 'MMBench_CN_V11',
+        }
+        ans_file2 = f'{root}/{mapp[dataset_name]}.tsv'
+        for f in [ans_file1, ans_file2]:
+            if osp.exists(f):
+                data = load(f)
+                if 'answer' in data and sum([pd.isna(x) for x in data['answer']]) == 0:
+                    return True
+    return False
+
+
+class NumpyEncoder(json.JSONEncoder):
+    def default(self, obj):
+        if isinstance(obj, (np.int_, np.intc, np.intp, np.int8,
+                            np.int16, np.int32, np.int64, np.uint8,
+                            np.uint16, np.uint32, np.uint64)):
+            return int(obj)
+        elif isinstance(obj, (np.float_, np.float16, np.float32, np.float64)):
+            return float(obj)
+        elif isinstance(obj, (np.complex_, np.complex64, np.complex128)):
+            return {'real': obj.real, 'imag': obj.imag}
+        elif isinstance(obj, (np.ndarray,)):
+            return obj.tolist()
+        elif isinstance(obj, (np.bool_)):
+            return bool(obj)
+        elif isinstance(obj, (np.void)):
+            return None
+        return json.JSONEncoder.default(self, obj)
+
+
+# LOAD & DUMP
+def dump(data, f, **kwargs):
+    def dump_pkl(data, pth, **kwargs):
+        pickle.dump(data, open(pth, 'wb'))
+
+    def dump_json(data, pth, **kwargs):
+        json.dump(data, open(pth, 'w'), indent=4, ensure_ascii=False, cls=NumpyEncoder)
+
+    def dump_jsonl(data, f, **kwargs):
+        lines = [json.dumps(x, ensure_ascii=False, cls=NumpyEncoder) for x in data]
+        with open(f, 'w', encoding='utf8') as fout:
+            fout.write('\n'.join(lines))
+
+    def dump_xlsx(data, f, **kwargs):
+        data.to_excel(f, index=False, engine='xlsxwriter')
+
+    def dump_csv(data, f, quoting=csv.QUOTE_ALL):
+        data.to_csv(f, index=False, encoding='utf-8', quoting=quoting)
+
+    def dump_tsv(data, f, quoting=csv.QUOTE_ALL):
+        data.to_csv(f, sep='\t', index=False, encoding='utf-8', quoting=quoting)
+
+    handlers = dict(pkl=dump_pkl, json=dump_json, jsonl=dump_jsonl, xlsx=dump_xlsx, csv=dump_csv, tsv=dump_tsv)
+    suffix = f.split('.')[-1]
+    return handlers[suffix](data, f, **kwargs)
+
+
+def load(f):
+    def load_pkl(pth):
+        return pickle.load(open(pth, 'rb'))
+
+    def load_json(pth):
+        return json.load(open(pth, 'r', encoding='utf-8'))
+
+    def load_jsonl(f):
+        lines = open(f, encoding='utf-8').readlines()
+        lines = [x.strip() for x in lines]
+        if lines[-1] == '':
+            lines = lines[:-1]
+        data = [json.loads(x) for x in lines]
+        return data
+
+    def load_xlsx(f):
+        return pd.read_excel(f)
+
+    def load_csv(f):
+        return pd.read_csv(f)
+
+    def load_tsv(f):
+        return pd.read_csv(f, sep='\t')
+
+    handlers = dict(pkl=load_pkl, json=load_json, jsonl=load_jsonl, xlsx=load_xlsx, csv=load_csv, tsv=load_tsv)
+    suffix = f.split('.')[-1]
+    return handlers[suffix](f)
+
+
+def download_file(url, filename=None):
+    import urllib.request
+    from tqdm import tqdm
+
+    class DownloadProgressBar(tqdm):
+        def update_to(self, b=1, bsize=1, tsize=None):
+            if tsize is not None:
+                self.total = tsize
+            self.update(b * bsize - self.n)
+
+    if filename is None:
+        filename = url.split('/')[-1]
+
+    with DownloadProgressBar(unit='B', unit_scale=True,
+                             miniters=1, desc=url.split('/')[-1]) as t:
+        urllib.request.urlretrieve(url, filename=filename, reporthook=t.update_to)
+    return filename
+
+
+def ls(dirname='.', match=[], mode='all', level=1):
+    if isinstance(level, str):
+        assert '+' in level
+        level = int(level[:-1])
+        res = []
+        for i in range(1, level + 1):
+            res.extend(ls(dirname, match=match, mode='file', level=i))
+        return res
+
+    if dirname == '.':
+        ans = os.listdir(dirname)
+    else:
+        ans = [osp.join(dirname, x) for x in os.listdir(dirname)]
+    assert mode in ['all', 'dir', 'file']
+    assert level >= 1 and isinstance(level, int)
+    if level == 1:
+        if isinstance(match, str):
+            match = [match]
+        for m in match:
+            if len(m) == 0:
+                continue
+            if m[0] != '!':
+                ans = [x for x in ans if m in x]
+            else:
+                ans = [x for x in ans if m[1:] not in x]
+        if mode == 'dir':
+            ans = [x for x in ans if osp.isdir(x)]
+        elif mode == 'file':
+            ans = [x for x in ans if not osp.isdir(x)]
+        return ans
+    else:
+        dirs = [x for x in ans if osp.isdir(x)]
+        res = []
+        for d in dirs:
+            res.extend(ls(d, match=match, mode=mode, level=level - 1))
+        return res
+
+
+def mrlines(fname, sp='\n'):
+    f = open(fname).read().split(sp)
+    while f != [] and f[-1] == '':
+        f = f[:-1]
+    return f
+
+
+def mwlines(lines, fname):
+    with open(fname, 'w') as fout:
+        fout.write('\n'.join(lines))
+
+
+def md5(s):
+    hash = hashlib.new('md5')
+    if osp.exists(s):
+        with open(s, 'rb') as f:
+            for chunk in iter(lambda: f.read(2**20), b''):
+                hash.update(chunk)
+    else:
+        hash.update(s.encode('utf-8'))
+    return str(hash.hexdigest())
+
+
+def last_modified(pth):
+    stamp = osp.getmtime(pth)
+    m_ti = time.ctime(stamp)
+    t_obj = time.strptime(m_ti)
+    t = time.strftime('%Y%m%d%H%M%S', t_obj)[2:]
+    return t
+
+
+def parse_file(s):
+    if osp.exists(s):
+        assert osp.isfile(s)
+        suffix = osp.splitext(s)[1].lower()
+        mime = mimetypes.types_map.get(suffix, 'unknown')
+        return (mime, s)
+    elif validators.url(s):
+        suffix = osp.splitext(s)[1].lower()
+        if suffix in mimetypes.types_map:
+            mime = mimetypes.types_map[suffix]
+            dname = osp.join(LMUDataRoot(), 'files')
+            os.makedirs(dname, exist_ok=True)
+            tgt = osp.join(dname, md5(s) + suffix)
+            download_file(s, tgt)
+            return (mime, tgt)
+        else:
+            return ('url', s)
+    else:
+        return (None, s)
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/smp/log.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/smp/log.py
new file mode 100644
index 0000000000000000000000000000000000000000..95804d5ed4048e6aca6535e1e0fde64525c0fac3
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/smp/log.py
@@ -0,0 +1,44 @@
+import logging
+
+logger_initialized = {}
+
+
+def get_logger(name, log_file=None, log_level=logging.INFO, file_mode='w'):
+    logger = logging.getLogger(name)
+    if name in logger_initialized:
+        return logger
+
+    for logger_name in logger_initialized:
+        if name.startswith(logger_name):
+            return logger
+
+    stream_handler = logging.StreamHandler()
+    handlers = [stream_handler]
+
+    try:
+        import torch.distributed as dist
+        if dist.is_available() and dist.is_initialized():
+            rank = dist.get_rank()
+        else:
+            rank = 0
+    except ImportError:
+        rank = 0
+
+    if rank == 0 and log_file is not None:
+        file_handler = logging.FileHandler(log_file, file_mode)
+        handlers.append(file_handler)
+
+    formatter = logging.Formatter(
+        '%(asctime)s - %(name)s - %(levelname)s - %(message)s')
+    for handler in handlers:
+        handler.setFormatter(formatter)
+        handler.setLevel(log_level)
+        logger.addHandler(handler)
+
+    if rank == 0:
+        logger.setLevel(log_level)
+    else:
+        logger.setLevel(logging.ERROR)
+
+    logger_initialized[name] = True
+    return logger
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/smp/misc.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/smp/misc.py
new file mode 100644
index 0000000000000000000000000000000000000000..7640fad28849c325e0955fdb9f43d27a1c9ecc64
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/smp/misc.py
@@ -0,0 +1,191 @@
+# flake8: noqa: F401, F403
+import abc
+import argparse
+import csv
+import multiprocessing as mp
+import os
+import os.path as osp
+import copy as cp
+import random as rd
+import requests
+import shutil
+import subprocess
+import warnings
+import logging
+import pandas as pd
+from collections import OrderedDict, defaultdict
+from multiprocessing import Pool, current_process
+from tqdm import tqdm
+import datetime
+import matplotlib.pyplot as plt
+import seaborn as sns
+from tabulate import tabulate_formats, tabulate
+from huggingface_hub import scan_cache_dir
+from sty import fg, bg, ef, rs
+
+def process_punctuation(inText):
+    import re
+    outText = inText
+    punct = [
+        ';', r'/', '[', ']', '"', '{', '}', '(', ')', '=', '+', '\\', '_', '-',
+        '>', '<', '@', '`', ',', '?', '!'
+    ]
+    commaStrip = re.compile('(\d)(,)(\d)')  # noqa: W605
+    periodStrip = re.compile('(?!<=\d)(\.)(?!\d)')  # noqa: W605
+    for p in punct:
+        if (p + ' ' in inText or ' ' + p in inText) or (re.search(
+                commaStrip, inText) is not None):
+            outText = outText.replace(p, '')
+        else:
+            outText = outText.replace(p, ' ')
+    outText = periodStrip.sub('', outText, re.UNICODE)
+    return outText
+
+def h2r(value):
+    if value[0] == '#':
+        value = value[1:]
+    assert len(value) == 6
+    return tuple(int(value[i:i + 2], 16) for i in range(0, 6, 2))
+
+def r2h(rgb):
+    return '#%02x%02x%02x' % rgb
+
+def colored(s, color):
+    if isinstance(color, str):
+        if hasattr(fg, color):
+            return getattr(fg, color) + s + fg.rs
+        color = h2r(color)
+    return fg(*color) + s + fg.rs
+
+def istype(s, type):
+    if isinstance(s, type):
+        return True
+    try:
+        return isinstance(eval(s), type)
+    except Exception as _:
+        return False
+
+def bincount(lst):
+    bins = defaultdict(lambda: 0)
+    for item in lst:
+        bins[item] += 1
+    return bins
+
+def get_cache_path(repo_id):
+    hf_cache_info = scan_cache_dir()
+    repos = list(hf_cache_info.repos)
+    repo = None
+    for r in repos:
+        if r.repo_id == repo_id:
+            repo = r
+            break
+    if repo is None:
+        return None
+    revs = list(repo.revisions)
+    rev2keep, last_modified = None, 0
+    for rev in revs:
+        if rev.last_modified > last_modified:
+            rev2keep, last_modified = rev, rev.last_modified
+    if rev2keep is None:
+        return None
+    return str(rev2keep.snapshot_path)
+
+def proxy_set(s):
+    import os
+    for key in ['http_proxy', 'HTTP_PROXY', 'https_proxy', 'HTTPS_PROXY']:
+        os.environ[key] = s
+
+def get_rank_and_world_size():
+    rank = int(os.environ.get('RANK', 0))
+    world_size = int(os.environ.get('WORLD_SIZE', 1))
+    return rank, world_size
+
+def splitlen(s, sym='/'):
+    return len(s.split(sym))
+
+def listinstr(lst, s):
+    assert isinstance(lst, list)
+    for item in lst:
+        if item in s:
+            return True
+    return False
+
+def d2df(D):
+    return pd.DataFrame({x: [D[x]] for x in D})
+
+def cn_string(s):
+    import re
+    if re.search(u'[\u4e00-\u9fff]', s):
+        return True
+    return False
+
+try:
+    import decord
+except ImportError:
+    pass
+
+def timestr(second=True, minute=False):
+    s = datetime.datetime.now().strftime('%Y%m%d%H%M%S')[2:]
+    if second:
+        return s
+    elif minute:
+        return s[:-2]
+    else:
+        return s[:-4]
+
+def dict_merge(dct, merge_dct):
+    for k, _ in merge_dct.items():
+        if (k in dct and isinstance(dct[k], dict) and isinstance(merge_dct[k], dict)):  #noqa
+            dict_merge(dct[k], merge_dct[k])
+        else:
+            dct[k] = merge_dct[k]
+
+def youtube_dl(idx):
+    cmd = f'youtube-dl -f best -f mp4 "{idx}"  -o {idx}.mp4'
+    os.system(cmd)
+
+def run_command(cmd):
+    if isinstance(cmd, str):
+        cmd = cmd.split()
+    return subprocess.check_output(cmd).decode()
+
+def load_env():
+    logger = logging.getLogger('LOAD_ENV')
+    try:
+        import vlmeval
+    except ImportError:
+        logger.error('VLMEval is not installed. Failed to import environment variables from .env file. ')
+        return
+    pth = osp.realpath(vlmeval.__path__[0])
+    pth = osp.join(pth, '../.env')
+    pth = osp.realpath(pth)
+    if not osp.exists(pth):
+        logger.error(f'Did not detect the .env file at {pth}, failed to load. ')
+        return
+
+    from dotenv import dotenv_values
+    values = dotenv_values(pth)
+    for k, v in values.items():
+        if v is not None and len(v):
+            os.environ[k] = v
+    logger.info(f'API Keys successfully loaded from {pth}')
+
+def pip_install_robust(package):
+    import sys
+    retry = 3
+    while retry > 0:
+        try:
+            package_base = package.split('=')[0]
+            module = __import__(package)
+            return True
+        except ImportError:
+            subprocess.check_call([sys.executable, '-m', 'pip', 'install', package])
+            retry -= 1
+    return False
+
+
+def version_cmp(v1, v2, op='eq'):
+    from packaging import version
+    import operator
+    op_func = getattr(operator, op)
+    return op_func(version.parse(v1), version.parse(v2))
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/smp/vlm.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/smp/vlm.py
new file mode 100644
index 0000000000000000000000000000000000000000..6577cec1a9292daa3a3dbe2a42b0f7bee892f9cf
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/smp/vlm.py
@@ -0,0 +1,134 @@
+import os
+import io
+import pandas as pd
+import numpy as np
+import string
+from uuid import uuid4
+import os.path as osp
+import base64
+from PIL import Image
+from .file import load, dump
+Image.MAX_IMAGE_PIXELS = 1e9
+
+
+def mmqa_display(question, target_size=512):
+    question = {k.lower(): v for k, v in question.items()}
+    keys = list(question.keys())
+    keys = [k for k in keys if k not in ['index', 'image']]
+
+    images = question['image']
+    if isinstance(images, str):
+        images = [images]
+
+    idx = question.pop('index', 'XXX')
+    print(f'INDEX: {idx}')
+
+    for im in images:
+        image = decode_base64_to_image(im, target_size=target_size)
+        display(image)  # noqa: F821
+
+    for k in keys:
+        try:
+            if not pd.isna(question[k]):
+                print(f'{k.upper()}. {question[k]}')
+        except ValueError:
+            if False in pd.isna(question[k]):
+                print(f'{k.upper()}. {question[k]}')
+
+
+def encode_image_to_base64(img, target_size=-1):
+    # if target_size == -1, will not do resizing
+    # else, will set the max_size ot (target_size, target_size)
+    if img.mode in ('RGBA', 'P'):
+        img = img.convert('RGB')
+    tmp = osp.join('/tmp', str(uuid4()) + '.jpg')
+    if target_size > 0:
+        img.thumbnail((target_size, target_size))
+    img.save(tmp)
+    with open(tmp, 'rb') as image_file:
+        image_data = image_file.read()
+    ret = base64.b64encode(image_data).decode('utf-8')
+    os.remove(tmp)
+    return ret
+
+
+def encode_image_file_to_base64(image_path, target_size=-1):
+    image = Image.open(image_path)
+    return encode_image_to_base64(image, target_size=target_size)
+
+
+def decode_base64_to_image(base64_string, target_size=-1):
+    image_data = base64.b64decode(base64_string)
+    image = Image.open(io.BytesIO(image_data))
+    if image.mode in ('RGBA', 'P'):
+        image = image.convert('RGB')
+    if target_size > 0:
+        image.thumbnail((target_size, target_size))
+    return image
+
+
+def decode_base64_to_image_file(base64_string, image_path, target_size=-1):
+    image = decode_base64_to_image(base64_string, target_size=target_size)
+    image.save(image_path)
+
+
+def build_option_str(option_dict):
+    s = 'There are several options: \n'
+    for c, content in option_dict.items():
+        if not pd.isna(content):
+            s += f'{c}. {content}\n'
+    return s
+
+
+def isimg(s):
+    return osp.exists(s) or s.startswith('http')
+
+
+def read_ok(img_path):
+    if not osp.exists(img_path):
+        return False
+    try:
+        im = Image.open(img_path)
+        assert im.size[0] > 0 and im.size[1] > 0
+        return True
+    except:
+        return False
+
+
+def gpt_key_set():
+    openai_key = os.environ.get('OPENAI_API_KEY', None)
+    return isinstance(openai_key, str) and openai_key.startswith('sk-')
+
+
+def apiok(wrapper):
+    s = wrapper.generate('Hello!')
+    return wrapper.fail_msg not in s
+
+
+def circular_pred(df, extract_func=None):
+    if extract_func is None:
+        extract_func = lambda x: x  # noqa: E731
+    df = df.sort_values('index')
+    from vlmeval.utils import can_infer_option
+    shift = int(1e6)
+
+    choices = [extract_func(x) for x in df['prediction']]
+    pred_map = {i: c for i, c in zip(df['index'], choices)}
+    flag_map = {i: True for i in pred_map if i < 1e6}
+    valid_map = {i: True for i in pred_map if i < 1e6}
+    for i in df['index']:
+        if i >= shift and pred_map[i] and pred_map[i - shift]:
+            if (
+                pred_map[i] not in list(string.ascii_uppercase) or  # noqa: W504
+                pred_map[i - shift] not in list(string.ascii_uppercase)
+            ):
+
+                valid_map[i % shift] = False
+                continue
+            if (ord(pred_map[i]) - ord(pred_map[i - shift])) % 4 == 1:
+                continue
+            else:
+                flag_map[i % shift] = False
+    flag_map = {k: v for k, v in flag_map.items() if valid_map[k]}
+    flags = list(flag_map.values())
+    return np.mean(flags)
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/utils/__init__.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/utils/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..930367891e263a35a6b537cf1c494c5dff299260
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/utils/__init__.py
@@ -0,0 +1,12 @@
+from .matching_util import can_infer, can_infer_option, can_infer_text
+from .mp_util import track_progress_rich
+from .custom_prompt import CustomPrompt
+from .dataset_config import dataset_URLs, img_root_map, DATASET_TYPE, abbr2full
+from .dataset import TSVDataset, split_MMMU, MMMU_result_transfer
+
+
+__all__ = [
+    'can_infer', 'can_infer_option', 'can_infer_text', 'track_progress_rich',
+    'TSVDataset', 'dataset_URLs', 'img_root_map', 'DATASET_TYPE', 'CustomPrompt',
+    'split_MMMU', 'abbr2full'
+]
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/utils/custom_prompt.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/utils/custom_prompt.py
new file mode 100644
index 0000000000000000000000000000000000000000..5841fbcec61217d4d2c61369455c3211a8ba8b45
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/utils/custom_prompt.py
@@ -0,0 +1,33 @@
+from ..smp import *
+from .dataset_config import img_root_map
+from abc import abstractmethod
+
+
+class CustomPrompt:
+
+    @abstractmethod
+    def use_custom_prompt(self, dataset):
+        raise NotImplementedError
+
+    @abstractmethod
+    def build_prompt(self, line, dataset):
+        raise NotImplementedError
+
+    def dump_image(self, line, dataset):
+        ROOT = LMUDataRoot()
+        assert isinstance(dataset, str)
+        img_root = osp.join(ROOT, 'images', img_root_map[dataset] if dataset in img_root_map else dataset)
+        os.makedirs(img_root, exist_ok=True)
+        if isinstance(line['image'], list):
+            tgt_path = []
+            assert 'image_path' in line
+            for img, im_name in zip(line['image'], line['image_path']):
+                path = osp.join(img_root, im_name)
+                if not read_ok(path):
+                    decode_base64_to_image_file(img, path)
+                tgt_path.append(path)
+        else:
+            tgt_path = osp.join(img_root, f"{line['index']}.jpg")
+            if not read_ok(tgt_path):
+                decode_base64_to_image_file(line['image'], tgt_path)
+        return tgt_path
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/utils/dataset.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/utils/dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..dbbb440e1353f4ae69d1067e2172544235069af5
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/utils/dataset.py
@@ -0,0 +1,190 @@
+import pandas as pd
+import hashlib
+from ..smp import *
+from .dataset_config import dataset_URLs, dataset_md5_dict, DATASET_TYPE
+from .custom_prompt import CustomPrompt
+from .matching_util import can_infer
+
+
+def isliststr(s):
+    return (s[0] == '[') and (s[-1] == ']')
+
+
+def check_md5(data_path, dataset):
+    if dataset not in dataset_md5_dict:
+        warnings.warn(f'We do not have an md5 record for dataset {dataset}, skip the md5 check. ')
+        return True
+    assert osp.exists(data_path)
+    with open(data_path, 'rb') as f:
+        hash = hashlib.new('md5')
+        for chunk in iter(lambda: f.read(2**20), b''):
+            hash.update(chunk)
+    if str(hash.hexdigest()) == dataset_md5_dict[dataset]:
+        return True
+    else:
+        warnings.warn('this data file is incomplete, so it needs to be downloaded again.')
+        return False
+
+
+def split_MMMU(msgs):
+    text, images = None, []
+    for s in msgs:
+        if s['type'] == 'image':
+            images.append(s['value'])
+        elif s['type'] == 'text':
+            assert text is None
+            text = s['value']
+    text_segs = text.split('<image ')
+    segs = [dict(type='text', value=text_segs[0])]
+    for i, seg in enumerate(text_segs):
+        if i == 0:
+            continue
+        assert istype(seg[0], int) and seg[1] == '>'
+        image_idx = int(seg[0]) - 1
+        segs.append(dict(type='image', value=images[image_idx]))
+        segs.append(dict(type='text', value=seg[2:]))
+    return segs
+
+
+def MMMU_result_transfer(result_path):
+    res = {}
+    result_data = load(result_path)
+    mcq = result_data['A'].notna()
+    lt = len(result_data)
+    for i in range(lt):
+        line = result_data.iloc[i]
+        if mcq[i]:
+            options = {
+                cand: line[cand]
+                for cand in string.ascii_uppercase
+                if cand in line and not pd.isna(line[cand])
+            }
+            prediction = line['prediction']
+            infer_prediction = can_infer(prediction, options)
+            res[line['id']] = infer_prediction
+        else:
+            res[line['id']] = line['prediction']
+    result_json = result_path.replace('.xlsx', '.json')
+    dump(res, result_json)
+    return result_json
+
+
+class TSVDataset(CustomPrompt):
+
+    def __init__(self, dataset='MMBench', skip_noimg=True):
+
+        self.data_root = LMUDataRoot()
+        assert osp.exists(self.data_root)
+
+        self.dataset = dataset
+        self.dataset_type = DATASET_TYPE(dataset)
+
+        if dataset in dataset_URLs:
+            url = dataset_URLs[dataset]
+            file_name = url.split('/')[-1]
+            data_path = osp.join(self.data_root, file_name)
+
+            if osp.exists(data_path) and check_md5(data_path, dataset):
+                pass
+            elif osp.isfile(url):
+            # If url is actually a file path, use it directly
+                data_path = url
+            else:
+                warnings.warn('The dataset tsv is not downloaded')
+                download_file(url, data_path)
+        else:
+            data_path = osp.join(self.data_root, dataset + '.tsv')
+            assert osp.exists(data_path)
+
+        data = load(data_path)
+        self.skip_noimg = skip_noimg
+        if skip_noimg and 'image' in data:
+            data = data[~pd.isna(data['image'])]
+
+        # Prompt for Captioning
+        if listinstr(['COCO'], dataset):
+            data['question'] = [(
+                'Please describe this image in general. Directly provide the description, '
+                'do not include prefix like "This image depicts". '
+            )] * len(data)
+
+        data['index'] = [str(x) for x in data['index']]
+
+        self.meta_only = True
+        if 'image' in data:
+            data['image'] = [str(x) for x in data['image']]
+
+            image_map = {x: y for x, y in zip(data['index'], data['image'])}
+            for k in image_map:
+                if len(image_map[k]) <= 64:
+                    idx = image_map[k]
+                    assert idx in image_map and len(image_map[idx]) > 64
+                    image_map[k] = image_map[idx]
+
+            data['image'] = [
+                eval(image_map[k]) if isliststr(image_map[k]) else image_map[k]
+                for k in data['index']
+            ]
+            self.meta_only = False
+
+        if 'image_path' in data:
+            data['image_path'] = [
+                eval(pths) if isliststr(pths) else pths for pths in data['image_path']
+            ]
+
+        if np.all([istype(x, int) for x in data['index']]):
+            data['index'] = [int(x) for x in data['index']]
+
+        self.data = data
+
+    def __len__(self):
+        return len(self.data)
+
+    def build_prompt(self, line, dataset=None):
+        if dataset is None:
+            dataset = self.dataset
+
+        if isinstance(line, int):
+            line = self.data.iloc[line]
+
+        if self.meta_only:
+            tgt_path = line['image_path']
+        else:
+            tgt_path = self.dump_image(line, dataset)
+
+        prompt = line['question']
+        if DATASET_TYPE(dataset) == 'multi-choice':
+            question = line['question']
+            options = {
+                cand: line[cand]
+                for cand in string.ascii_uppercase
+                if cand in line and not pd.isna(line[cand])
+            }
+            options_prompt = 'Options:\n'
+            for key, item in options.items():
+                options_prompt += f'{key}. {item}\n'
+            hint = line['hint'] if ('hint' in line and not pd.isna(line['hint'])) else None
+            prompt = ''
+            if hint is not None:
+                prompt += f'Hint: {hint}\n'
+            prompt += f'Question: {question}\n'
+            if len(options):
+                prompt += options_prompt
+                prompt += 'Please select the correct answer from the options above. \n'
+        elif DATASET_TYPE(dataset) == 'VQA':
+            if listinstr(['ocrvqa', 'textvqa', 'chartqa', 'docvqa'], dataset.lower()):
+                prompt += '\nPlease try to answer the question with short words or phrases if possible\n.'
+
+        msgs = []
+        if isinstance(tgt_path, list):
+            msgs.extend([dict(type='image', value=p) for p in tgt_path])
+        else:
+            msgs = [dict(type='image', value=tgt_path)]
+        msgs.append(dict(type='text', value=prompt))
+
+        return msgs
+
+    def display(self, line):
+        if isinstance(line, int):
+            line = self.data.iloc[line]
+        mmqa_display(line)
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/utils/dataset_config.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/utils/dataset_config.py
new file mode 100644
index 0000000000000000000000000000000000000000..933665317a6ff3b386822bac80791b3325dacf25
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/utils/dataset_config.py
@@ -0,0 +1,158 @@
+from ..smp import listinstr
+
+dataset_URLs = {
+    # MMBench v1.0
+    'MMBench_DEV_EN': 'https://opencompass.openxlab.space/utils/VLMEval/MMBench_DEV_EN.tsv',
+    'MMBench_TEST_EN': 'https://opencompass.openxlab.space/utils/VLMEval/MMBench_TEST_EN.tsv',
+    'MMBench_DEV_CN': 'https://opencompass.openxlab.space/utils/VLMEval/MMBench_DEV_CN.tsv',
+    'MMBench_TEST_CN': 'https://opencompass.openxlab.space/utils/VLMEval/MMBench_TEST_CN.tsv',
+    'MMBench': 'https://opencompass.openxlab.space/utils/VLMEval/MMBench.tsv',  # Internal Only
+    'MMBench_CN': 'https://opencompass.openxlab.space/utils/VLMEval/MMBench_CN.tsv',    # Internal Only
+    # MMBench v1.1
+    'MMBench_DEV_EN_V11': 'https://opencompass.openxlab.space/utils/VLMEval/MMBench_DEV_EN_V11.tsv',
+    'MMBench_TEST_EN_V11': 'https://opencompass.openxlab.space/utils/VLMEval/MMBench_TEST_EN_V11.tsv',
+    'MMBench_DEV_CN_V11': 'https://opencompass.openxlab.space/utils/VLMEval/MMBench_DEV_CN_V11.tsv',
+    'MMBench_TEST_CN_V11': 'https://opencompass.openxlab.space/utils/VLMEval/MMBench_TEST_CN_V11.tsv',
+    'MMBench_V11': 'https://opencompass.openxlab.space/utils/VLMEval/MMBench_V11.tsv',  # Internal Only
+    'MMBench_CN_V11': 'https://opencompass.openxlab.space/utils/VLMEval/MMBench_CN_V11.tsv',    # Internal Only
+    # CCBench
+    'CCBench': 'https://opencompass.openxlab.space/utils/VLMEval/CCBench.tsv',
+    'MME': 'https://opencompass.openxlab.space/utils/VLMEval/MME.tsv',
+    'SEEDBench_IMG': 'https://opencompass.openxlab.space/utils/VLMEval/SEEDBench_IMG.tsv',
+    'CORE_MM': 'https://opencompass.openxlab.space/utils/VLMEval/CORE_MM.tsv',
+    'MMVet': 'https://opencompass.openxlab.space/utils/VLMEval/MMVet.tsv',
+    'COCO_VAL': 'https://opencompass.openxlab.space/utils/VLMEval/COCO_VAL.tsv',
+    'OCRVQA_TEST': 'https://opencompass.openxlab.space/utils/VLMEval/OCRVQA_TEST.tsv',
+    'OCRVQA_TESTCORE': 'https://opencompass.openxlab.space/utils/VLMEval/OCRVQA_TESTCORE.tsv',
+    'TextVQA_VAL': 'https://opencompass.openxlab.space/utils/VLMEval/TextVQA_VAL.tsv',
+    'MMMU_DEV_VAL': 'https://opencompass.openxlab.space/utils/VLMEval/MMMU_DEV_VAL.tsv',
+    'MMMU_TEST': 'https://opencompass.openxlab.space/utils/VLMEval/MMMU_TEST.tsv',
+    'MathVista_MINI': 'https://opencompass.openxlab.space/utils/VLMEval/MathVista_MINI.tsv',
+    'ScienceQA_VAL': 'https://opencompass.openxlab.space/utils/VLMEval/ScienceQA_VAL.tsv',
+    'ScienceQA_TEST': 'https://opencompass.openxlab.space/utils/VLMEval/ScienceQA_TEST.tsv',
+    'HallusionBench': 'https://opencompass.openxlab.space/utils/VLMEval/HallusionBench.tsv',
+    'DocVQA_VAL': 'https://opencompass.openxlab.space/utils/VLMEval/DocVQA_VAL.tsv',
+    'DocVQA_TEST': 'https://opencompass.openxlab.space/utils/VLMEval/DocVQA_TEST.tsv',
+    'InfoVQA_VAL': 'https://opencompass.openxlab.space/utils/VLMEval/InfoVQA_VAL.tsv',
+    'InfoVQA_TEST': 'https://opencompass.openxlab.space/utils/VLMEval/InfoVQA_TEST.tsv',
+    'AI2D_TEST': 'https://opencompass.openxlab.space/utils/VLMEval/AI2D_TEST.tsv',
+    'LLaVABench': 'https://opencompass.openxlab.space/utils/VLMEval/LLaVABench.tsv',
+    'OCRBench': 'https://opencompass.openxlab.space/utils/VLMEval/OCRBench.tsv',
+    'ChartQA_TEST': 'https://opencompass.openxlab.space/utils/VLMEval/ChartQA_TEST.tsv',
+    'MMStar': 'https://opencompass.openxlab.space/utils/VLMEval/MMStar.tsv',
+    'RealWorldQA': 'https://opencompass.openxlab.space/utils/VLMEval/RealWorldQA.tsv',
+    'POPE': 'https://opencompass.openxlab.space/utils/VLMEval/POPE.tsv',
+}
+
+dataset_md5_dict = {
+    # MMBench v1.0
+    'MMBench_DEV_EN': 'b6caf1133a01c6bb705cf753bb527ed8',
+    'MMBench_TEST_EN': '6939fadb0ce626fefc0bdc9c64efc528',
+    'MMBench_DEV_CN': '08b8fc3324a5ed74155350f57be69fbd',
+    'MMBench_TEST_CN': '7e1239baf0ee4c8b513e19705a0f317e',
+    'MMBench': '4115aea3383f3dd0083be6a633e0f820',  # Internal Only
+    'MMBench_CN': '2e053ffc90ea598b1feae13c36dc13ee',    # Internal Only
+    # MMBench v1.1
+    'MMBench_DEV_EN_V11': '30c05be8f2f347a50be25aa067248184',
+    'MMBench_TEST_EN_V11': '26f0f15381a21720255091d3e0316ce6',
+    'MMBench_DEV_CN_V11': '593f9b5f6bea453d870a798b34ae4f37',
+    'MMBench_TEST_CN_V11': '74bbe4556dac745613c7cbe5ad787050',
+    'MMBench_V11': 'b9276414f57af1308dcc4d0cd9b42e7c',  # Internal Only
+    'MMBench_CN_V11': '95f6980dd1b4de38e3cbffe0305a3f25',    # Internal Only
+    # CCBench
+    'CCBench': '1de88b4257e7eee3f60b18d45eda6f07',
+    'MME': 'b36b43c3f09801f5d368627fb92187c3',
+    'SEEDBench_IMG': '68017231464752261a2526d6ca3a10c0',
+    'CORE_MM': '8a8da2f2232e79caf98415bfdf0a202d',
+    'MMVet': '748aa6d4aa9d4de798306a63718455e3',
+    'COCO_VAL': '72a5079dead060269ac222c5aa5128af',
+    'OCRVQA_TEST': 'ca46a6d74b403e9d6c0b670f6fc00db9',
+    'OCRVQA_TESTCORE': 'c5239fe77db8bdc1f2ad8e55e0d1fe97',
+    'TextVQA_VAL': 'b233b31f551bbf4056f2f955da3a92cd',
+    'MMMU_DEV_VAL': '521afc0f3bf341e6654327792781644d',
+    'MMMU_TEST': 'c19875d11a2d348d07e5eb4bdf33166d',
+    'MathVista_MINI': 'f199b98e178e5a2a20e7048f5dcb0464',
+    'ScienceQA_VAL': '96320d05e142e585e7204e72affd29f3',
+    'ScienceQA_TEST': 'e42e9e00f9c59a80d8a5db35bc32b71f',
+    'HallusionBench': '0c23ac0dc9ef46832d7a24504f2a0c7c',
+    'DocVQA_VAL': 'd5ee77e1926ff10690d469c56b73eabf',
+    'DocVQA_TEST': '6a2f28cac26ef2d3447374e8c6f6c8e9',
+    'InfoVQA_VAL': '2342e9c225222f0ef4dec545ebb126fe',
+    'InfoVQA_TEST': 'df535bf51b88dc9718252c34131a6227',
+    'AI2D_TEST': '0f593e0d1c7df9a3d69bf1f947e71975',
+    'LLaVABench': 'd382a093f749a697820d3dadd61c8428',
+    'OCRBench': 'e953d98a987cc6e26ef717b61260b778',
+    'ChartQA_TEST': 'c902e0aa9be5582a7aad6dcf52734b42',
+    'MMStar': 'e1ecd2140806c1b1bbf54b43372efb9e',
+    'RealWorldQA': '92321028d2bc29040284b6674721e48f',
+    'POPE': 'c12f5acb142f2ef1f85a26ba2fbe41d5',
+}
+
+img_root_map = {k: k for k in dataset_URLs}
+img_root_map.update({
+    # MMBench v1.0
+    'MMBench_DEV_EN': 'MMBench',
+    'MMBench_TEST_EN': 'MMBench',
+    'MMBench_DEV_CN': 'MMBench',
+    'MMBench_TEST_CN': 'MMBench',
+    'MMBench': 'MMBench',   # Internal Only
+    'MMBench_CN': 'MMBench',    # Internal Only
+    # MMBench v1.1
+    'MMBench_DEV_EN_V11': 'MMBench_V11',
+    'MMBench_TEST_EN_V11': 'MMBench_V11',
+    'MMBench_DEV_CN_V11': 'MMBench_V11',
+    'MMBench_TEST_CN_V11': 'MMBench_V11',
+    'MMBench_V11': 'MMBench_V11',   # Internal Only
+    'MMBench_CN_V11': 'MMBench_V11',    # Internal Only
+    'COCO_VAL': 'COCO',
+    'OCRVQA_TEST': 'OCRVQA',
+    'OCRVQA_TESTCORE': 'OCRVQA',
+    'TextVQA_VAL': 'TextVQA',
+    'MMMU_DEV_VAL': 'MMMU',
+    'MMMU_TEST': 'MMMU',
+    'MathVista_MINI': 'MathVista',
+    'HallusionBench': 'Hallusion',
+    'DocVQA_VAL': 'DocVQA',
+    'DocVQA_TEST': 'DocVQA_TEST',
+    'OCRBench': 'OCRBench',
+    'ChartQA_TEST': 'ChartQA_TEST',
+    'InfoVQA_VAL': 'InfoVQA_VAL',
+    'InfoVQA_TEST': 'InfoVQA_TEST',
+    'MMStar': 'MMStar',
+    'RealWorldQA': 'RealWorldQA',
+    'POPE': 'POPE',
+})
+
+assert set(dataset_URLs) == set(img_root_map)
+
+
+def DATASET_TYPE(dataset):
+    # Dealing with Custom Dataset
+    dataset = dataset.lower()
+    if listinstr(['mmbench', 'seedbench', 'ccbench', 'mmmu', 'scienceqa', 'ai2d', 'mmstar', 'realworldqa'], dataset):
+        return 'multi-choice'
+    elif listinstr(['mme', 'hallusion', 'pope'], dataset):
+        return 'Y/N'
+    elif 'coco' in dataset:
+        return 'Caption'
+    elif listinstr(['ocrvqa', 'textvqa', 'chartqa', 'mathvista', 'docvqa', 'infovqa', 'llavabench',
+                    'mmvet', 'ocrbench'], dataset):
+        return 'VQA'
+    else:
+        if dataset not in dataset_URLs:
+            import warnings
+            warnings.warn(f"Dataset {dataset} not found in dataset_URLs, will use 'multi-choice' as the default TYPE.")
+            return 'multi-choice'
+        else:
+            return 'QA'
+
+
+def abbr2full(s):
+    datasets = [x for x in img_root_map]
+    ins = [s in d for d in datasets]
+    if sum(ins) == 1:
+        for d in datasets:
+            if s in d:
+                return d
+    else:
+        return s
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/utils/matching_util.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/utils/matching_util.py
new file mode 100644
index 0000000000000000000000000000000000000000..ccdf1b7d563125a4763724b81cf0164b06b94185
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/utils/matching_util.py
@@ -0,0 +1,69 @@
+import string
+import copy as cp
+import os
+from ..smp import *
+
+
+def can_infer_option(answer, choices):
+    verbose = os.environ.get('VERBOSE', 0)
+    # Choices is a dictionary
+    if 'Failed to obtain answer via API' in answer:
+        return False
+
+    reject_to_answer = [
+        "Sorry, I can't help with images of people yet.",
+        "I can't process this file.",
+        "I'm sorry, but without the image provided",
+        'Cannot determine the answer'
+    ]
+    for err in reject_to_answer:
+        if err in answer:
+            return 'Z'
+
+    def count_choice(splits, choices, prefix='', suffix=''):
+        cnt = 0
+        for c in choices:
+            if prefix + c + suffix in splits:
+                cnt += 1
+        return cnt
+
+    answer_mod = cp.copy(answer)
+    chars = '.()[],:;!*#{}'
+    for c in chars:
+        answer_mod = answer_mod.replace(c, ' ')
+
+    splits = [x.strip() for x in answer_mod.split()]
+    count = count_choice(splits, choices)
+
+    if count == 1:
+        for ch in choices:
+            if 'A' in splits and len(splits) > 3 and verbose:
+                logger = get_logger('Evaluation')
+                logger.info(f'A might be a quantifier in the string: {answer}.')
+                return False
+            if ch in splits:
+                return ch
+    elif count == 0 and count_choice(splits, {'Z', ''}) == 1:
+        return 'Z'
+    return False
+
+
+def can_infer_text(answer, choices):
+    answer = answer.lower()
+    assert isinstance(choices, dict)
+    for k in choices:
+        assert k in string.ascii_uppercase
+        choices[k] = str(choices[k]).lower()
+    cands = []
+    for k in choices:
+        if choices[k] in answer:
+            cands.append(k)
+    if len(cands) == 1:
+        return cands[0]
+    return False
+
+
+def can_infer(answer, choices):
+    answer = str(answer)
+    copt = can_infer_option(answer, choices)
+    return copt if copt else can_infer_text(answer, choices)
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/utils/mp_util.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/utils/mp_util.py
new file mode 100644
index 0000000000000000000000000000000000000000..2322e2d0e404cb632f4954e6a08ac386c6d38a11
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/utils/mp_util.py
@@ -0,0 +1,191 @@
+from multiprocessing import Pool
+import os
+from typing import Callable, Iterable, Sized
+
+from rich.progress import (BarColumn, MofNCompleteColumn, Progress, Task,
+                           TaskProgressColumn, TextColumn, TimeRemainingColumn)
+from rich.text import Text
+import os.path as osp
+import portalocker
+from ..smp import load, dump
+
+
+class _Worker:
+    """Function wrapper for ``track_progress_rich``"""
+
+    def __init__(self, func) -> None:
+        self.func = func
+
+    def __call__(self, inputs):
+        inputs, idx = inputs
+        if not isinstance(inputs, (tuple, list, dict)):
+            inputs = (inputs, )
+
+        if isinstance(inputs, dict):
+            return self.func(**inputs), idx
+        else:
+            return self.func(*inputs), idx
+
+
+class _SkipFirstTimeRemainingColumn(TimeRemainingColumn):
+    """Skip calculating remaining time for the first few times.
+
+    Args:
+        skip_times (int): The number of times to skip. Defaults to 0.
+    """
+
+    def __init__(self, *args, skip_times=0, **kwargs):
+        super().__init__(*args, **kwargs)
+        self.skip_times = skip_times
+
+    def render(self, task: Task) -> Text:
+        """Show time remaining."""
+        if task.completed <= self.skip_times:
+            return Text('-:--:--', style='progress.remaining')
+        return super().render(task)
+
+
+def _tasks_with_index(tasks):
+    """Add index to tasks."""
+    for idx, task in enumerate(tasks):
+        yield task, idx
+
+
+def track_progress_rich(func: Callable,
+                        tasks: Iterable = tuple(),
+                        task_num: int = None,
+                        nproc: int = 1,
+                        chunksize: int = 1,
+                        description: str = 'Processing',
+                        save=None, keys=None,
+                        color: str = 'blue') -> list:
+    """Track the progress of parallel task execution with a progress bar. The
+    built-in :mod:`multiprocessing` module is used for process pools and tasks
+    are done with :func:`Pool.map` or :func:`Pool.imap_unordered`.
+
+    Args:
+        func (callable): The function to be applied to each task.
+        tasks (Iterable or Sized): A tuple of tasks. There are several cases
+            for different format tasks:
+            - When ``func`` accepts no arguments: tasks should be an empty
+              tuple, and ``task_num`` must be specified.
+            - When ``func`` accepts only one argument: tasks should be a tuple
+              containing the argument.
+            - When ``func`` accepts multiple arguments: tasks should be a
+              tuple, with each element representing a set of arguments.
+              If an element is a ``dict``, it will be parsed as a set of
+              keyword-only arguments.
+            Defaults to an empty tuple.
+        task_num (int, optional): If ``tasks`` is an iterator which does not
+            have length, the number of tasks can be provided by ``task_num``.
+            Defaults to None.
+        nproc (int): Process (worker) number, if nuproc is 1,
+            use single process. Defaults to 1.
+        chunksize (int): Refer to :class:`multiprocessing.Pool` for details.
+            Defaults to 1.
+        description (str): The description of progress bar.
+            Defaults to "Process".
+        color (str): The color of progress bar. Defaults to "blue".
+
+    Examples:
+        >>> import time
+
+        >>> def func(x):
+        ...    time.sleep(1)
+        ...    return x**2
+        >>> track_progress_rich(func, range(10), nproc=2)
+
+    Returns:
+        list: The task results.
+    """
+    if save is not None:
+        assert osp.exists(osp.dirname(save)) or osp.dirname(save) == ''
+        if not osp.exists(save):
+            dump({}, save)
+    if keys is not None:
+        assert len(keys) == len(tasks)
+
+    if not callable(func):
+        raise TypeError('func must be a callable object')
+    if not isinstance(tasks, Iterable):
+        raise TypeError(
+            f'tasks must be an iterable object, but got {type(tasks)}')
+    if isinstance(tasks, Sized):
+        if len(tasks) == 0:
+            if task_num is None:
+                raise ValueError('If tasks is an empty iterable, '
+                                 'task_num must be set')
+            else:
+                tasks = tuple(tuple() for _ in range(task_num))
+        else:
+            if task_num is not None and task_num != len(tasks):
+                raise ValueError('task_num does not match the length of tasks')
+            task_num = len(tasks)
+
+    if nproc <= 0:
+        raise ValueError('nproc must be a positive number')
+
+    skip_times = nproc * chunksize if nproc > 1 else 0
+    prog_bar = Progress(
+        TextColumn('{task.description}'),
+        BarColumn(),
+        _SkipFirstTimeRemainingColumn(skip_times=skip_times),
+        MofNCompleteColumn(),
+        TaskProgressColumn(show_speed=True),
+    )
+
+    worker = _Worker(func)
+    task_id = prog_bar.add_task(
+        total=task_num, color=color, description=description)
+    tasks = _tasks_with_index(tasks)
+
+    # Use single process when nproc is 1, else use multiprocess.
+    with prog_bar:
+        if nproc == 1:
+            results = []
+            for task in tasks:
+                result, idx = worker(task)
+                results.append(worker(task)[0])
+                if save is not None:
+                    with portalocker.Lock(save, timeout=5) as fh:
+                        ans = load(save)
+                        ans[keys[idx]] = result
+
+                        if os.environ.get('VERBOSE', True):
+                            print(keys[idx], result, flush=True)
+
+                        dump(ans, save)
+                        fh.flush()
+                        os.fsync(fh.fileno())
+
+                prog_bar.update(task_id, advance=1, refresh=True)
+        else:
+            with Pool(nproc) as pool:
+                results = []
+                unordered_results = []
+                gen = pool.imap_unordered(worker, tasks, chunksize)
+                try:
+                    for result in gen:
+                        result, idx = result
+                        unordered_results.append((result, idx))
+
+                        if save is not None:
+                            with portalocker.Lock(save, timeout=5) as fh:
+                                ans = load(save)
+                                ans[keys[idx]] = result
+
+                                if os.environ.get('VERBOSE', False):
+                                    print(keys[idx], result, flush=True)
+
+                                dump(ans, save)
+                                fh.flush()
+                                os.fsync(fh.fileno())
+
+                        results.append(None)
+                        prog_bar.update(task_id, advance=1, refresh=True)
+                except Exception as e:
+                    prog_bar.stop()
+                    raise e
+            for result, idx in unordered_results:
+                results[idx] = result
+    return results
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/vlm/__init__.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/vlm/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..51b2b2b6ac2106839421526daf74a15970ab38c2
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/vlm/__init__.py
@@ -0,0 +1,7 @@
+import torch
+
+torch.set_grad_enabled(False)
+torch.manual_seed(1234)
+from .base import BaseModel
+from .minicpm_llama3_v_2_5 import MiniCPM_Llama3_V
+from .minicpm_v import MiniCPM_V
\ No newline at end of file
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/vlm/base.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/vlm/base.py
new file mode 100644
index 0000000000000000000000000000000000000000..70a5b7bfd10a9f1a2797b8e9b194297c6cf0da8a
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/vlm/base.py
@@ -0,0 +1,150 @@
+from ..smp import *
+from ..utils.dataset_config import img_root_map
+from abc import abstractmethod
+
+
+class BaseModel:
+
+    INTERLEAVE = False
+    allowed_types = ['text', 'image']
+
+    def use_custom_prompt(self, dataset):
+        """Whether to use custom prompt for the given dataset.
+
+        Args:
+            dataset (str): The name of the dataset.
+
+        Returns:
+            bool: Whether to use custom prompt. If True, will call `build_prompt` of the VLM to build the prompt.
+                Default to False.
+        """
+        return False
+
+    @abstractmethod
+    def build_prompt(self, line, dataset):
+        """Build custom prompts for a specific dataset. Called only if `use_custom_prompt` returns True.
+
+        Args:
+            line (line of pd.DataFrame): The raw input line.
+            dataset (str): The name of the dataset.
+
+        Returns:
+            str: The built message.
+        """
+        raise NotImplementedError
+
+    def dump_image(self, line, dataset):
+        """Dump the image(s) of the input line to the corresponding dataset folder.
+
+        Args:
+            line (line of pd.DataFrame): The raw input line.
+            dataset (str): The name of the dataset.
+
+        Returns:
+            str | list[str]: The paths of the dumped images.
+        """
+        ROOT = LMUDataRoot()
+        assert isinstance(dataset, str)
+        img_root = osp.join(ROOT, 'images', img_root_map[dataset] if dataset in img_root_map else dataset)
+        os.makedirs(img_root, exist_ok=True)
+        if isinstance(line['image'], list):
+            tgt_path = []
+            assert 'image_path' in line
+            for img, im_name in zip(line['image'], line['image_path']):
+                path = osp.join(img_root, im_name)
+                if not read_ok(path):
+                    decode_base64_to_image_file(img, path)
+                tgt_path.append(path)
+        else:
+            tgt_path = osp.join(img_root, f"{line['index']}.jpg")
+            if not read_ok(tgt_path):
+                decode_base64_to_image_file(line['image'], tgt_path)
+            tgt_path = [tgt_path]
+        return tgt_path
+
+    @abstractmethod
+    def generate_inner(self, message, dataset=None):
+        raise NotImplementedError
+
+    def check_content(self, msgs):
+        """Check the content type of the input. Four types are allowed: str, dict, liststr, listdict.
+        """
+        if isinstance(msgs, str):
+            return 'str'
+        if isinstance(msgs, dict):
+            return 'dict'
+        if isinstance(msgs, list):
+            types = [self.check_content(m) for m in msgs]
+            if all(t == 'str' for t in types):
+                return 'liststr'
+            if all(t == 'dict' for t in types):
+                return 'listdict'
+        return 'unknown'
+
+    def preproc_content(self, inputs):
+        """Convert the raw input messages to a list of dicts.
+
+        Args:
+            inputs: raw input messages.
+
+        Returns:
+            list(dict): The preprocessed input messages. Will return None if failed to preprocess the input.
+        """
+        if self.check_content(inputs) == 'str':
+            return [dict(type='text', value=inputs)]
+        elif self.check_content(inputs) == 'dict':
+            assert 'type' in inputs and 'value' in inputs
+            return [inputs]
+        elif self.check_content(inputs) == 'liststr':
+            res = []
+            for s in inputs:
+                mime, pth = parse_file(s)
+                if mime is None or mime == 'unknown':
+                    res.append(dict(type='text', value=s))
+                else:
+                    res.append(dict(type=mime.split('/')[0], value=pth))
+            return res
+        elif self.check_content(inputs) == 'listdict':
+            for item in inputs:
+                assert 'type' in item and 'value' in item
+                mime, s = parse_file(item['value'])
+                if mime is None:
+                    assert item['type'] == 'text'
+                else:
+                    assert mime.split('/')[0] == item['type']
+                    item['value'] = s
+            return inputs
+        else:
+            return None
+
+    def generate(self, message, dataset=None):
+        """Generate the output message.
+
+        Args:
+            message (list[dict]): The input message.
+            dataset (str, optional): The name of the dataset. Defaults to None.
+
+        Returns:
+            str: The generated message.
+        """
+        assert self.check_content(message) in ['str', 'dict', 'liststr', 'listdict'], f'Invalid input type: {message}'
+        message = self.preproc_content(message)
+        assert message is not None and self.check_content(message) == 'listdict'
+        for item in message:
+            assert item['type'] in self.allowed_types, f'Invalid input type: {item["type"]}'
+        return self.generate_inner(message, dataset)
+
+    def message_to_promptimg(self, message):
+        assert not self.INTERLEAVE
+        model_name = self.__class__.__name__
+        warnings.warn(
+            f'Model {model_name} does not support interleaved input. '
+            'Will use the first image and aggregated texts as prompt. ')
+        num_images = len([x for x in message if x['type'] == 'image'])
+        if num_images == 0:
+            prompt = '\n'.join([x['value'] for x in message if x['type'] == 'text'])
+            image = None
+        else:
+            prompt = '\n'.join([x['value'] for x in message if x['type'] == 'text'])
+            image = [x['value'] for x in message if x['type'] == 'image'][0]
+        return prompt, image
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/vlm/minicpm_llama3_v_2_5.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/vlm/minicpm_llama3_v_2_5.py
new file mode 100644
index 0000000000000000000000000000000000000000..a0d06adaeec55c76794b000f35167e146a0ed8de
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/vlm/minicpm_llama3_v_2_5.py
@@ -0,0 +1,155 @@
+import torch
+from PIL import Image
+from transformers import AutoModel, AutoTokenizer
+
+from ..smp import *
+from ..utils import DATASET_TYPE
+from .base import BaseModel
+
+
+class MiniCPM_Llama3_V(BaseModel):
+    INSTALL_REQ = False
+    INTERLEAVE = True
+
+    def __init__(self, model_path='openbmb/MiniCPM-V', **kwargs):
+        assert model_path is not None
+        self.model_path = model_path
+        self.ckpt = model_path
+
+        print(f'load from {self.model_path}')
+        self.model = AutoModel.from_pretrained(self.model_path, trust_remote_code=True)
+        if '.pt' in model_path:
+            print(f'load from {model_path}')
+            self.state_dict = torch.load(self.ckpt, map_location='cpu')
+            self.model.load_state_dict(self.state_dict, strict=False)
+        self.model = self.model.to(dtype=torch.float16)
+        self.model.eval().cuda()
+        self.kwargs = kwargs
+        self.tokenizer = AutoTokenizer.from_pretrained(self.model_path, trust_remote_code=True)
+        torch.cuda.empty_cache()
+        self.num_beams = 1 if self.model_path == 'openbmb/MiniCPM-V' else 3
+        self.options_system_prompt = ('Carefully read the following question and select the letter corresponding '
+                                      'to the correct answer. Highlight the applicable choices without giving '
+                                      'explanations.')
+        self.wo_options_system_prompt = 'Carefully read the following question Answer the question directly.'
+        self.detail_system_prompt = 'Answer this question in detail.'
+        self.vqa_prompt = 'Answer the question using a single word or phrase.'
+        
+
+    def use_custom_prompt(self, dataset):
+        if listinstr(['multi-choice', 'VQA'], DATASET_TYPE(dataset)):
+            return True
+        elif dataset is not None and listinstr(['HallusionBench'], dataset):
+            return True
+        return False
+
+    def build_prompt(self, line, dataset=None):
+        if dataset is None:
+            dataset = self.dataset
+
+        if isinstance(line, int):
+            line = self.data.iloc[line]
+
+        tgt_path = self.dump_image(line, dataset)
+        system_prompt = ''
+
+        question = line['question']
+        if DATASET_TYPE(dataset) == 'multi-choice':
+            options = {
+                cand: line[cand]
+                for cand in string.ascii_uppercase
+                if cand in line and not pd.isna(line[cand])
+            }
+            options_prompt = 'Options:\n'
+            for key, item in options.items():
+                options_prompt += f'{key}. {item}\n'
+            hint = line['hint'] if ('hint' in line and not pd.isna(line['hint'])) else None
+            prompt = ''
+            if hint is not None:
+                prompt += f'Hint: {hint}\n'
+            prompt += f'Question: {question}\n'
+            if len(options):
+                prompt += options_prompt
+                system_prompt = self.options_system_prompt +  "\nPlease just indicate your choice."
+            else:
+                system_prompt = self.wo_options_system_prompt
+
+            if 'MMMU' in dataset: # Corner Case
+                prompt = system_prompt + '\n' + prompt
+                system_prompt = ''
+        elif dataset is not None and listinstr(['HallusionBench'], dataset):
+            question = line['question'] + " Yes or No?"
+            prompt = question
+        
+        elif dataset is not None and listinstr(['OCRBench'], dataset):
+            system_prompt = self.vqa_prompt
+            question = line['question']
+            prompt = question
+        elif DATASET_TYPE(dataset) == 'VQA':
+            if listinstr(['LLaVABench'], dataset):
+                system_prompt = ""
+                prompt = question
+            elif listinstr(['MMVet'], dataset):
+                system_prompt = self.detail_system_prompt
+                prompt = question
+            else:
+                system_prompt = self.vqa_prompt
+                prompt = question
+
+        msgs = []
+        if system_prompt:
+            msgs.append(dict(type='text', value=system_prompt))
+        if isinstance(tgt_path, list):
+            msgs.extend([dict(type='image', value=p) for p in tgt_path])
+        else:
+            msgs = [dict(type='image', value=tgt_path)]
+        msgs.append(dict(type='text', value=prompt))
+
+        return msgs
+
+    def generate_inner(self, message, dataset=None):
+        if DATASET_TYPE(dataset) == 'multi-choice':
+            max_new_tokens = 200
+        elif DATASET_TYPE(dataset) == 'Y/N':
+            max_new_tokens = 3
+        else:
+            max_new_tokens = 1024
+
+        '''
+        nums_beams = 3
+        '''
+        default_kwargs = dict(
+            max_new_tokens=max_new_tokens,
+            sampling=False,
+            num_beams=self.num_beams,
+        )
+        default_kwargs.update(self.kwargs)
+
+        content = []
+        
+    #     message = [
+    #     {'type': 'text', 'value': 'sys prompt'},
+    #     {'type': 'image', 'value': '/path/to/image1.jpg'},
+    #     {'type': 'text', 'value': 'Here is an image:'},
+    # ]
+
+        for x in message:
+            if x['type'] == 'text':
+                content.append(x['value'])
+            elif x['type'] == 'image':
+                image = Image.open(x['value']).convert('RGB')
+                content.append(image)
+        msgs = [{'role': 'user', 'content': content}]
+
+        res = self.model.chat(
+            image = None,
+            msgs=msgs,
+            context=None,
+            tokenizer=self.tokenizer,
+            **default_kwargs
+        )
+
+        if isinstance(res, tuple) and len(res) > 0:
+            res = res[0]
+        # print(f"content: {content}, res: {res}")
+        return res
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/vlm/minicpm_v.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/vlm/minicpm_v.py
new file mode 100644
index 0000000000000000000000000000000000000000..7605482a45de6707598e066984a5c0cb4833287b
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vlmevalkit/vlmeval/vlm/minicpm_v.py
@@ -0,0 +1,85 @@
+import torch
+from PIL import Image
+from transformers import AutoModel, AutoTokenizer
+
+from .base import BaseModel
+from ..smp import *
+from ..utils import DATASET_TYPE
+
+
+class MiniCPM_V(BaseModel):
+
+    INSTALL_REQ = False
+    INTERLEAVE = False
+
+    def __init__(self, model_path='openbmb/MiniCPM-V', **kwargs):
+        assert model_path is not None
+        self.model_path = model_path
+        print(f'load from {self.model_path}')
+        self.model = AutoModel.from_pretrained(self.model_path, trust_remote_code=True)
+        self.model = self.model.to(dtype=torch.bfloat16)
+        self.model.eval().cuda()
+        self.kwargs = kwargs
+        self.tokenizer = AutoTokenizer.from_pretrained(self.model_path, trust_remote_code=True)
+        torch.cuda.empty_cache()
+        self.num_beams = 1 if self.model_path == 'openbmb/MiniCPM-V' else 3
+
+    def use_custom_prompt(self, dataset):
+        assert dataset is not None
+        if listinstr(['MMMU'], dataset):
+            return True
+        return False
+
+    def build_prompt(self, line, dataset=None):
+        assert dataset is None or isinstance(dataset, str)
+        assert self.use_custom_prompt(dataset)
+        tgt_path = self.dump_image(line, dataset)
+
+        question = line['question']
+        options = {
+            cand: line[cand]
+            for cand in string.ascii_uppercase
+            if cand in line and not pd.isna(line[cand])
+        }
+        options_prompt = 'Options:\n'
+        for key, item in options.items():
+            options_prompt += f'{key}. {item}\n'
+        hint = line['hint'] if ('hint' in line and not pd.isna(line['hint'])) else None
+        prompt = ''
+        if hint is not None:
+            prompt += f'Hint: {hint}\n'
+        prompt += f'{question}\n'
+        if len(options):
+            prompt += options_prompt
+            prompt = 'Study the image carefully and pick the option associated with the correct answer. \
+                Focus solely on selecting the option and avoid including any other content.\n' + prompt
+        message = [dict(type='text', value=prompt)]
+        message.extend([dict(type='image', value=p) for p in tgt_path])
+
+        return message
+
+    def generate_inner(self, message, dataset=None):
+        prompt, image_path = self.message_to_promptimg(message)
+        image = Image.open(image_path).convert('RGB')
+        msgs = [{'role': 'user', 'content': prompt}]
+        if DATASET_TYPE(dataset) == 'multi-choice':
+            max_new_tokens = 20
+        elif DATASET_TYPE(dataset) == 'Y/N':
+            max_new_tokens = 100
+        else:
+            max_new_tokens = 1024
+
+        default_kwargs = dict(
+            max_new_tokens=max_new_tokens,
+            sampling=False,
+            num_beams=self.num_beams
+        )
+        default_kwargs.update(self.kwargs)
+        res, _, _ = self.model.chat(
+            image=image,
+            msgs=msgs,
+            context=None,
+            tokenizer=self.tokenizer,
+            **default_kwargs
+        )
+        return res
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/README.md b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..b6fd3f4f3886751f838c8708e029e6ff717ab4fe
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/README.md
@@ -0,0 +1,3 @@
+# vqa-eval
+
+contains vqa_eval kit from the server.
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/datasets/__init__.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/datasets/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/datasets/vqa_dataset.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/datasets/vqa_dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..83c9e9aef5359ebed23b85448b382119d8488736
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/datasets/vqa_dataset.py
@@ -0,0 +1,116 @@
+import json
+import os
+import re
+from torch.utils.data import Dataset
+
+def prompt_processor(prompt):
+    if prompt.startswith('OCR tokens: '):
+        pattern = r"Question: (.*?) Short answer:"
+        match = re.search(pattern, prompt, re.DOTALL)
+        question = match.group(1)
+    elif 'Reference OCR token: ' in prompt and len(prompt.split('\n')) == 3:
+        if prompt.startswith('Reference OCR token:'):
+            question = prompt.split('\n')[1]
+        else:
+            question = prompt.split('\n')[0]
+    elif len(prompt.split('\n')) == 2:
+        question = prompt.split('\n')[0]
+    else:
+        assert False
+
+    return question.lower()
+    
+class textVQADataset(Dataset):
+    def __init__(
+        self,
+        image_dir="./downloads/TextVQA/train_images",
+        ann_path="./downloads/TextVQA/TextVQA_0.5.1_val.json",
+    ):
+        self.data = json.load(open(ann_path, "r"))["data"]
+        self.image_dir = image_dir
+
+    def __len__(self):
+        return len(self.data)
+
+    def __getitem__(self, idx):
+        question = self.data[idx]['question']
+        answers = self.data[idx]['answers']
+        img_id = self.data[idx]['image_id']
+        qid = self.data[idx]['question_id']
+        img_path = os.path.join(self.image_dir, f"{img_id}.jpg")
+        
+        item = {
+            "question_id": qid,
+            "image_path": img_path,
+            "question": question,
+            "gt_answers": answers
+        }
+        
+        return item
+    
+class docVQADataset(Dataset):
+    def __init__(
+        self,
+        image_dir= "./downloads/DocVQA/spdocvqa_images",
+        ann_path= "./downloads/DocVQA/val_v1.0_withQT.json",
+        ocr_token_path=None
+    ):
+
+        self.data = json.load(open(ann_path, "r"))["data"]
+        self.image_dir = image_dir
+        self.ann_path = ann_path
+        if ocr_token_path:
+            self.ocr_token_data = {item['image_id']: item for item in json.load(open(ocr_token_path, "r"))["data"]}
+
+    def __len__(self):
+        return len(self.data)
+
+    def __getitem__(self, idx):
+        question_id = self.data[idx]['questionId']  
+        relative_img_path = self.data[idx]['image']
+        corrected_relative_img_path = relative_img_path.replace("documents", "images")
+        img_path = os.path.join(self.image_dir, corrected_relative_img_path)
+        question = self.data[idx]['question']
+        answers = self.data[idx]['answers']
+        
+        question_type = self.data[idx]['question_types']
+        
+        return {
+            "question_id": question_id,  
+            "image_path": img_path,
+            "question": question,
+            "gt_answers": answers,
+            'question_type': question_type,
+        }
+
+
+class docVQATESTDataset(Dataset):
+    def __init__(
+        self,
+        image_dir= "./downloads/DocVQA/spdocvqa_images",
+        ann_path= "./downloads/DocVQA/test_v1.0.json",
+        ocr_token_path=None
+    ):
+
+        self.data = json.load(open(ann_path, "r"))["data"]
+        self.image_dir = image_dir
+        self.ann_path = ann_path
+
+    def __len__(self):
+        return len(self.data)
+
+    def __getitem__(self, idx):
+        question_id = self.data[idx]['questionId']  
+        relative_img_path = self.data[idx]['image']
+        corrected_relative_img_path = relative_img_path.replace("documents", "images")
+        img_path = os.path.join(self.image_dir, corrected_relative_img_path)
+        question = self.data[idx]['question']
+        
+        
+        return {
+            "question_id": question_id,  
+            "image_path": img_path,
+            "question": question,
+            "gt_answers": "",
+            'question_type': "",
+        }
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/eval.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/eval.py
new file mode 100644
index 0000000000000000000000000000000000000000..ee2bf68ef6c668bb865c277a935b1cef4753400e
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/eval.py
@@ -0,0 +1,97 @@
+import sys
+import datetime
+import json
+import os
+import torch
+
+script_dir = os.path.dirname(os.path.realpath(__file__))
+
+sys.path.append(os.path.join(script_dir, '..'))
+
+from datasets.vqa_dataset import docVQADataset, docVQATESTDataset, textVQADataset
+
+
+print(torch.__version__)
+
+import numpy as np
+
+from eval_utils.getargs import parse_args
+from eval_utils.vqa_evaluate import *
+
+
+def get_model(args):
+    if args.model_name=='':
+        raise Exception('Model name cannot be empty str!')
+    from models.MiniCPM.minicpmv import MiniCPM_V
+    model_path = args.model_path
+    ckpt = args.ckpt
+    model = MiniCPM_V(model_path=model_path, ckpt=ckpt, device=args.device)
+    
+    return model
+
+
+def main(args):
+    np.random.seed(0)
+    max_sample_num = None
+
+    torch.distributed.init_process_group(
+        backend='nccl',
+        world_size=int(os.getenv('WORLD_SIZE', '1')),
+        rank=int(os.getenv('RANK', '0')),
+    )
+    torch.cuda.set_device(int(os.getenv('LOCAL_RANK', 0)))
+    print(f'Init Rank-{torch.distributed.get_rank()}')
+    if torch.distributed.is_initialized():
+        args.device = torch.device(f"cuda:{torch.cuda.current_device()}")
+
+    model = get_model(args)
+    
+    result = {}
+    time = datetime.datetime.now().strftime("%Y%m%d%H%M%S")
+
+    if args.eval_textVQA or args.eval_all:
+        dataset = textVQADataset(args.textVQA_image_dir, args.textVQA_ann_path)
+        if max_sample_num is not None:
+            dataset = torch.utils.data.Subset(dataset, range(max_sample_num))
+        acc = evaluate_VQA(model, dataset, args.model_name, 'textVQA', time, \
+                batch_size=args.batchsize, generate_method=args.generate_method, answer_path=args.answer_path)
+        result['textVQA'] = acc
+
+    if args.eval_docVQA or args.eval_all:
+        dataset = docVQADataset(args.docVQA_image_dir, args.docVQA_ann_path)
+        if max_sample_num is not None:
+            dataset = torch.utils.data.Subset(dataset, range(max_sample_num))
+        acc = evaluate_VQA(model, dataset, args.model_name, 'docVQA', time, batch_size=args.batchsize, generate_method=args.generate_method, answer_path=args.answer_path)
+        result['docVQA'] = acc
+
+    if args.eval_docVQATest or args.eval_all:
+        target_dataset = "docVQATest"
+        dataset = docVQATESTDataset(args.docVQATest_image_dir, args.docVQATest_ann_path)
+        if max_sample_num is not None:
+            dataset = torch.utils.data.Subset(dataset, range(max_sample_num))
+        acc = evaluate_VQA(model, dataset, args.model_name, target_dataset, time, batch_size=args.batchsize, generate_method=args.generate_method, answer_path=args.answer_path)
+        result['docVQATest'] = acc
+    
+    if torch.distributed.is_initialized():
+        torch.distributed.barrier()
+
+    if torch.distributed.is_initialized() and torch.distributed.get_rank() != 0:
+        return None
+
+    result_path = os.path.join(os.path.join(args.answer_path, args.model_name), 'result.json')
+    
+    output_flag = False
+    for k, v in result.items():
+        if v > 0.0:
+            output_flag = True
+            break
+    
+    if output_flag:
+        with open(result_path, "w") as f:
+            f.write(json.dumps(result, indent=4))
+
+
+if __name__ == "__main__":
+    args = parse_args()
+
+    main(args)
\ No newline at end of file
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/eval_utils/cal_metric.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/eval_utils/cal_metric.py
new file mode 100644
index 0000000000000000000000000000000000000000..e5fb43986c47c40014eb47275dff927b49f21d44
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/eval_utils/cal_metric.py
@@ -0,0 +1,40 @@
+import json
+import glob
+import re
+
+def has_word(sentence, word):
+    pattern = r"\b" + re.escape(word) + r"\b"
+    match = re.search(pattern, sentence)
+    if match:
+        return True
+    else:
+        return False
+def remove_special_chars(s):
+    pattern = r"[^a-zA-Z0-9\s]"
+    s = re.sub(pattern, "", s)
+    return s
+
+for model in glob.glob('./answer_save/*'):
+    print(model, ':')
+    result_list = sorted(glob.glob(f'{model}/*.json'))
+    for task_result_path in result_list:
+        taskname = task_result_path.split('/')[-1]
+        taskname = taskname.split('.')[0]
+        if taskname not in ['IIIT5K', 'svt', 'IC13_857', 'IC15_1811', 'svtp', 'ct80',
+                            'cocotext', 'ctw', 'totaltext', 'HOST']:
+            continue
+
+        correct = 0
+        num = 0
+        with open(task_result_path, 'r') as f:
+            dict = json.load(f)[:100]
+            for i in range(len(dict)):
+                gt_answers = dict[i]['gt_answers']
+                answer = dict[i]['answer']
+                gt_answers = remove_special_chars(gt_answers).lower()
+                answer = remove_special_chars(answer).lower()
+                if has_word(answer, gt_answers):
+                    correct+=1
+                num+=1
+        print(f'{taskname:10s}:{float(correct)/num*100:.2f}')
+    print('=' * 32)
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/eval_utils/getargs.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/eval_utils/getargs.py
new file mode 100644
index 0000000000000000000000000000000000000000..42f74e8c6ab32426da936f611e2049ba3a0ff9de
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/eval_utils/getargs.py
@@ -0,0 +1,62 @@
+import argparse
+
+
+def parse_args():
+    parser = argparse.ArgumentParser(description="Demo")
+
+    parser.add_argument('--local-rank', type=int, default=0, help='Local rank for distributed training')
+
+    # textVQA
+    parser.add_argument("--textVQA_image_dir", type=str, default="")
+    parser.add_argument("--textVQA_ann_path", type=str, default="")
+
+    # docVQA
+    parser.add_argument("--docVQA_image_dir", type=str, default="")
+    parser.add_argument("--docVQA_ann_path", type=str, default="")
+
+    # docVQATest
+    parser.add_argument("--docVQATest_image_dir", type=str, default="")
+    parser.add_argument("--docVQATest_ann_path", type=str, default="")
+
+    # result path
+    parser.add_argument("--answer_path", type=str, default="./answers-new")
+
+    # eval
+    parser.add_argument(
+        "--eval_textVQA",
+        action="store_true",
+        default=False,
+        help="Whether to evaluate on textVQA."
+    )
+    parser.add_argument(
+        "--eval_docVQA",
+        action="store_true",
+        default=False,
+        help="Whether to evaluate on docVQA."
+    )
+    parser.add_argument(
+        "--eval_docVQATest",
+        action="store_true",
+        default=False,
+        help="Whether to evaluate on docVQA."
+    )
+    
+    parser.add_argument(
+        "--eval_all",
+        action="store_true",
+        default=False,
+        help="Whether to evaluate all datasets"
+    )
+
+    parser.add_argument("--model_name", type=str, default="")
+    parser.add_argument("--model_path", type=str, default="")
+
+    parser.add_argument("--generate_method", type=str, default="", help="generate with interleave or not.")
+
+    parser.add_argument("--device", type=str, default="cuda:0")
+    parser.add_argument('--batchsize', type=int, default=1, help='Batch size for processing.')
+
+    parser.add_argument("--ckpt", type=str, default=None)
+
+    args = parser.parse_args()
+    return args
\ No newline at end of file
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/eval_utils/vqa_evaluate.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/eval_utils/vqa_evaluate.py
new file mode 100644
index 0000000000000000000000000000000000000000..78888949ce171b972fd459d4e3a8fa37ec173692
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/eval_utils/vqa_evaluate.py
@@ -0,0 +1,446 @@
+import itertools
+import json
+import os
+import re
+from collections import namedtuple
+
+import torch
+from tqdm import tqdm
+
+
+class InferenceSampler(torch.utils.data.sampler.Sampler):
+
+    def __init__(self, size):
+        self._size = int(size)
+        assert size > 0
+        self._rank = torch.distributed.get_rank()
+        self._world_size = torch.distributed.get_world_size()
+        self._local_indices = self._get_local_indices(size, self._world_size,
+                                                      self._rank)
+
+    @staticmethod
+    def _get_local_indices(total_size, world_size, rank):
+        shard_size = total_size // world_size
+        left = total_size % world_size
+        shard_sizes = [shard_size + int(r < left) for r in range(world_size)]
+
+        begin = sum(shard_sizes[:rank])
+        end = min(sum(shard_sizes[:rank + 1]), total_size)
+        return range(begin, end)
+
+    def __iter__(self):
+        yield from self._local_indices
+
+    def __len__(self):
+        return len(self._local_indices)
+    
+def collate_fn_vqa(batches):
+    '''
+    '''
+    image_paths = [_['image_path'] for _ in batches]
+    questions = [_['question'] for _ in batches]
+    gt_answers = [_['gt_answers'] for _ in batches]
+    ocr_tokens = [_['ocr_tokens'] if 'ocr_tokens' in _ else None for _ in batches]
+    question_ids = [_['question_id'] if 'question_id' in _ else None for _ in batches]
+    question_type = [_['question_type'] if 'question_type' in _ else None for _ in batches]
+
+    return image_paths, questions, gt_answers, ocr_tokens, question_ids, question_type
+
+def has_word(sentence, word):
+    if word[0].isalnum():
+        start_pattern = r"\b"
+    else:
+        start_pattern = r""
+
+    if word[-1].isalnum():
+        end_pattern = r"\b"
+    else:
+        end_pattern = r""
+
+    pattern = start_pattern + re.escape(word) + end_pattern
+    match = re.search(pattern, sentence)
+    return bool(match)
+
+def remove_special_chars(s):
+    pattern = r"[^a-zA-Z0-9\s]"
+    s = re.sub(pattern, "", s)
+    return s
+
+def levenshtein_distance(s1, s2):
+    if len(s1) > len(s2):
+        s1, s2 = s2, s1
+
+    distances = range(len(s1) + 1)
+    for i2, c2 in enumerate(s2):
+        distances_ = [i2+1]
+        for i1, c1 in enumerate(s1):
+            if c1 == c2:
+                distances_.append(distances[i1])
+            else:
+                distances_.append(1 + min((distances[i1], distances[i1 + 1], distances_[-1])))
+        distances = distances_
+    return distances[-1]
+
+class VQAEval:
+    def __init__(self):
+        self.contractions = {
+            "aint": "ain't",
+            "arent": "aren't",
+            "cant": "can't",
+            "couldve": "could've",
+            "couldnt": "couldn't",
+            "couldn'tve": "couldn't've",
+            "couldnt've": "couldn't've",
+            "didnt": "didn't",
+            "doesnt": "doesn't",
+            "dont": "don't",
+            "hadnt": "hadn't",
+            "hadnt've": "hadn't've",
+            "hadn'tve": "hadn't've",
+            "hasnt": "hasn't",
+            "havent": "haven't",
+            "hed": "he'd",
+            "hed've": "he'd've",
+            "he'dve": "he'd've",
+            "hes": "he's",
+            "howd": "how'd",
+            "howll": "how'll",
+            "hows": "how's",
+            "Id've": "I'd've",
+            "I'dve": "I'd've",
+            "Im": "I'm",
+            "Ive": "I've",
+            "isnt": "isn't",
+            "itd": "it'd",
+            "itd've": "it'd've",
+            "it'dve": "it'd've",
+            "itll": "it'll",
+            "let's": "let's",
+            "maam": "ma'am",
+            "mightnt": "mightn't",
+            "mightnt've": "mightn't've",
+            "mightn'tve": "mightn't've",
+            "mightve": "might've",
+            "mustnt": "mustn't",
+            "mustve": "must've",
+            "neednt": "needn't",
+            "notve": "not've",
+            "oclock": "o'clock",
+            "oughtnt": "oughtn't",
+            "ow's'at": "'ow's'at",
+            "'ows'at": "'ow's'at",
+            "'ow'sat": "'ow's'at",
+            "shant": "shan't",
+            "shed've": "she'd've",
+            "she'dve": "she'd've",
+            "she's": "she's",
+            "shouldve": "should've",
+            "shouldnt": "shouldn't",
+            "shouldnt've": "shouldn't've",
+            "shouldn'tve": "shouldn't've",
+            "somebody'd": "somebodyd",
+            "somebodyd've": "somebody'd've",
+            "somebody'dve": "somebody'd've",
+            "somebodyll": "somebody'll",
+            "somebodys": "somebody's",
+            "someoned": "someone'd",
+            "someoned've": "someone'd've",
+            "someone'dve": "someone'd've",
+            "someonell": "someone'll",
+            "someones": "someone's",
+            "somethingd": "something'd",
+            "somethingd've": "something'd've",
+            "something'dve": "something'd've",
+            "somethingll": "something'll",
+            "thats": "that's",
+            "thered": "there'd",
+            "thered've": "there'd've",
+            "there'dve": "there'd've",
+            "therere": "there're",
+            "theres": "there's",
+            "theyd": "they'd",
+            "theyd've": "they'd've",
+            "they'dve": "they'd've",
+            "theyll": "they'll",
+            "theyre": "they're",
+            "theyve": "they've",
+            "twas": "'twas",
+            "wasnt": "wasn't",
+            "wed've": "we'd've",
+            "we'dve": "we'd've",
+            "weve": "we've",
+            "werent": "weren't",
+            "whatll": "what'll",
+            "whatre": "what're",
+            "whats": "what's",
+            "whatve": "what've",
+            "whens": "when's",
+            "whered": "where'd",
+            "wheres": "where's",
+            "whereve": "where've",
+            "whod": "who'd",
+            "whod've": "who'd've",
+            "who'dve": "who'd've",
+            "wholl": "who'll",
+            "whos": "who's",
+            "whove": "who've",
+            "whyll": "why'll",
+            "whyre": "why're",
+            "whys": "why's",
+            "wont": "won't",
+            "wouldve": "would've",
+            "wouldnt": "wouldn't",
+            "wouldnt've": "wouldn't've",
+            "wouldn'tve": "wouldn't've",
+            "yall": "y'all",
+            "yall'll": "y'all'll",
+            "y'allll": "y'all'll",
+            "yall'd've": "y'all'd've",
+            "y'alld've": "y'all'd've",
+            "y'all'dve": "y'all'd've",
+            "youd": "you'd",
+            "youd've": "you'd've",
+            "you'dve": "you'd've",
+            "youll": "you'll",
+            "youre": "you're",
+            "youve": "you've",
+        }
+        self.manualMap = {
+            "none": "0",
+            "zero": "0",
+            "one": "1",
+            "two": "2",
+            "three": "3",
+            "four": "4",
+            "five": "5",
+            "six": "6",
+            "seven": "7",
+            "eight": "8",
+            "nine": "9",
+            "ten": "10",
+        }
+        self.articles = ["a", "an", "the"]
+
+        self.periodStrip = re.compile("(?!<=\d)(\.)(?!\d)")
+        self.commaStrip = re.compile("(\d)(\,)(\d)")
+        self.punct = [
+            ";",
+            r"/",
+            "[",
+            "]",
+            '"',
+            "{",
+            "}",
+            "(",
+            ")",
+            "=",
+            "+",
+            "\\",
+            "_",
+            "-",
+            ">",
+            "<",
+            "@",
+            "`",
+            ",",
+            "?",
+            "!",
+        ]
+    def clean_text(self, text):
+        text = text.replace("\n", " ").replace("\t", " ").strip()
+        text = self.processPunctuation(text)
+        text = self.processDigitArticle(text)
+        return text
+    
+    def evaluate_vqa_human(self, answer, gt_answers):
+        '''TextVQA, VQAv2, OKVQA, vizwiz'''
+        answer = answer.replace("\n", " ").replace("\t", " ").strip()
+        answer = self.processPunctuation(answer)
+        answer = self.processDigitArticle(answer)
+        gt_answers = [self.processPunctuation(ans) for ans in gt_answers]
+        gt_answers = [self.processDigitArticle(ans) for ans in gt_answers]
+
+        gtAcc = [] 
+
+        for idx, gtAnsDatum in enumerate(gt_answers):  
+            otherGTAns = gt_answers[:idx] + gt_answers[idx+1:]
+
+            matchingAns = [item for item in otherGTAns if answer == item]
+            
+            acc = min(1, float(len(matchingAns)) / 3)
+            gtAcc.append(acc) 
+
+        avgGTAcc = float(sum(gtAcc)) / len(gtAcc) if gtAcc else 0  
+
+        return avgGTAcc
+
+    def evaluate_anls(self, answer, gt_answers, threshold=0.5):
+        '''DOcVQA, InfographicsVQA, STVQA'''
+        answer = ' '.join(answer.strip().lower().split())
+        if not isinstance(gt_answers, list):
+            gt_answers = [gt_answers]
+        gt_answers = [' '.join(gt_answer.strip().lower().split()) for gt_answer in gt_answers]
+
+        values = []
+        for gt_answer in gt_answers:
+            dist = levenshtein_distance(answer, gt_answer)
+            length = max(len(answer), len(gt_answer))
+            values.append(0.0 if length == 0 else float(dist) / float(length))
+
+        score = 1 - min(values)
+        
+        score = 0 if score < threshold else score
+        
+        return score
+
+    def processPunctuation(self, inText):
+        outText = inText
+        for p in self.punct:
+            if (p + " " in inText or " " + p in inText) or (
+                re.search(self.commaStrip, inText) != None
+            ):
+                outText = outText.replace(p, "")
+            else:
+                outText = outText.replace(p, " ")
+        outText = self.periodStrip.sub("", outText, re.UNICODE)
+        return outText
+
+    def processDigitArticle(self, inText):
+        outText = []
+        tempText = inText.lower().split()
+        for word in tempText:
+            word = self.manualMap.setdefault(word, word)
+            if word not in self.articles:
+                outText.append(word)
+            else:
+                pass
+        for wordId, word in enumerate(outText):
+            if word in self.contractions:
+                outText[wordId] = self.contractions[word]
+        outText = " ".join(outText)
+        return outText
+
+
+def evaluate_dataset(dataset_name, answer_file_path, model_name, method = None):
+    with open(answer_file_path, 'r', encoding='utf-8') as f:
+        predictions = json.load(f)
+
+    eval = VQAEval()
+    total_accuracy = 0
+    num = 0
+    Entry = namedtuple('Entry', ['text', 'bbox'])
+
+    for item in predictions:
+        gt_answers = item['gt_answers']
+        answer = item['answer']
+        if method is not None:
+            pass
+        if dataset_name in ["textVQA"]:
+            if num == 0:
+                print(f"evaluating vqa...")
+            accuracy = eval.evaluate_vqa_human(answer, gt_answers)
+        elif dataset_name in ['docVQA']:
+            if  num == 0:
+                print(f"evaluating anls...")
+            accuracy = eval.evaluate_anls(answer, gt_answers)
+        else:
+            accuracy = eval.evaluate_has(answer, gt_answers)
+        item['accuracy'] = accuracy
+
+        total_accuracy += accuracy
+        num += 1
+
+    average_accuracy = total_accuracy / num
+    print(f'{dataset_name}:{average_accuracy}')
+    
+    answer_model_method_path = answer_file_path.replace('.json', f'_{model_name}_{method}.json')
+    with open(answer_model_method_path, "w", encoding='utf-8') as f:
+        json.dump(predictions, f, indent=4, ensure_ascii=False)
+
+    return average_accuracy
+
+
+def evaluate_VQA(
+    model,
+    dataset,
+    model_name,
+    dataset_name,
+    time,
+    batch_size=1,
+    generate_method="interleave",
+    answer_path='./answers',
+):
+    print(f"answer path:{answer_path}")
+
+    sampler = None
+    if torch.distributed.is_initialized():
+        sampler=InferenceSampler(len(dataset))
+
+    dataloader = torch.utils.data.DataLoader(
+        dataset=dataset,
+        batch_size=batch_size,
+        sampler=sampler,
+        collate_fn=collate_fn_vqa
+    )
+    
+    now_rank = torch.distributed.get_rank()
+    
+    answer_dir = os.path.join(answer_path, model_name, time)
+    os.makedirs(answer_dir, exist_ok=True)
+    
+    image_list = []
+    for item in dataset:
+        image_list.append(item["image_path"])
+    
+    predictions = []
+    
+    for batch in tqdm(dataloader, desc="Running inference"):
+        image_paths, questions, gt_answers, ocr_tokens_list, question_ids, question_type  = batch
+
+        with torch.no_grad():
+            if model_name != "minicpm":
+                if model_name != "codellama":
+                    outputs = model.generate(images=image_paths, questions=questions, datasetname=dataset_name)
+                else:
+                    outputs = model.generate()
+            elif model_name == "minicpm":
+                if generate_method == "old":
+                    outputs = model.generate(images=image_paths, questions=questions, datasetname=dataset_name)
+                elif generate_method == "interleave":
+                    outputs = model.generate_with_interleaved(images=image_paths, questions=questions, datasetname=dataset_name)
+                else:
+                    raise Exception(f"Wrong generate paradigm {generate_method}!")
+            
+            for i in range(len(outputs)):
+                answer_dict = {
+                    'question_id': question_ids[i],
+                    'question': questions[i],
+                    'answer': outputs[i],
+                    'gt_answers': gt_answers[i],
+                    'image_path': image_paths[i],
+                    'model_name': model_name,
+                    'question_type': question_type[i]
+                }              
+                predictions.append(answer_dict)
+                    
+    if torch.distributed.is_initialized():
+        torch.distributed.barrier()
+    if torch.distributed.is_initialized():
+        world_size = torch.distributed.get_world_size()
+        merged_predictions = [None for _ in range(world_size)]
+        torch.distributed.all_gather_object(merged_predictions, predictions)
+        predictions = [_ for _ in itertools.chain.from_iterable(merged_predictions)]
+
+    if torch.distributed.is_initialized() and torch.distributed.get_rank() != 0:
+        return None
+    
+    answer_file_path = os.path.join(answer_dir, f"{dataset_name}.json")
+    print(f"answer_file_path:{answer_file_path}")
+    
+    with open(answer_file_path, "w", encoding='utf-8') as f:
+        json.dump(predictions, f, indent=4, ensure_ascii=False)
+
+    if dataset_name in ["docVQATest"]:
+        return -1.0
+
+    return evaluate_dataset(answer_file_path=answer_file_path, dataset_name=dataset_name, model_name=model_name)
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/models/MiniCPM/minicpmv.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/models/MiniCPM/minicpmv.py
new file mode 100644
index 0000000000000000000000000000000000000000..c6bdc696c1ef5438a198d365461928975046f914
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/models/MiniCPM/minicpmv.py
@@ -0,0 +1,97 @@
+
+import torch
+from PIL import Image
+from transformers import AutoModel, AutoTokenizer
+
+Image.MAX_IMAGE_PIXELS = 1000000000
+
+max_token  = {
+    'docVQA': 100,
+    'textVQA': 100,
+    "docVQATest": 100
+}
+
+class MiniCPM_V:
+
+    def __init__(self, model_path, ckpt, device=None)->None:
+        self.model_path = model_path
+        self.ckpt = ckpt
+        self.model = AutoModel.from_pretrained(self.model_path, trust_remote_code=True).eval()
+        if self.ckpt is not None:
+            self.ckpt = ckpt
+            self.state_dict = torch.load(self.ckpt, map_location=torch.device('cpu'))
+            self.model.load_state_dict(self.state_dict)
+            
+        self.model = self.model.to(dtype=torch.float16)
+        self.model.to(device)
+        
+        self.tokenizer = AutoTokenizer.from_pretrained(self.model_path, trust_remote_code=True)
+        torch.cuda.empty_cache()
+
+    def generate(self, images, questions, datasetname):
+        image = Image.open(images[0]).convert('RGB')
+        try:
+            max_new_tokens = max_token[datasetname]
+        except:
+            max_new_tokens = 1024
+        if (datasetname == 'docVQA') or (datasetname == "docVQATest") :
+            prompt = "Answer the question directly with single word." + "\n" + questions[0]
+        elif (datasetname == 'textVQA') :
+            prompt = "Answer the question directly with single word." + '\n'+ questions[0]
+        
+        msgs = [{'role': 'user', 'content': prompt}]
+        default_kwargs = dict(
+            max_new_tokens=max_new_tokens,
+            sampling=False,
+            num_beams=3
+        )
+        res = self.model.chat(
+            image=image,
+            msgs=msgs,
+            context=None,
+            tokenizer=self.tokenizer,
+            **default_kwargs
+        )
+        
+        return [res]
+    
+    def generate_with_interleaved(self, images, questions, datasetname):
+        try:
+            max_new_tokens = max_token[datasetname]
+        except:
+            max_new_tokens = 1024
+        
+        prompt = "Answer the question directly with single word."
+        
+        default_kwargs = dict(
+            max_new_tokens=max_new_tokens,
+            sampling=False,
+            num_beams=3
+        )
+        
+        content = []
+        message = [
+            {'type': 'text', 'value': prompt},
+            {'type': 'image', 'value': images[0]},
+            {'type': 'text', 'value': questions[0]}
+        ]
+        for x in message:
+            if x['type'] == 'text':
+                content.append(x['value'])
+            elif x['type'] == 'image':
+                image = Image.open(x['value']).convert('RGB')
+                content.append(image)
+        msgs = [{'role': 'user', 'content': content}]
+
+        res = self.model.chat(
+            image=None,
+            msgs=msgs,
+            context=None,
+            tokenizer=self.tokenizer,
+            **default_kwargs
+        )
+
+        if isinstance(res, tuple) and len(res) > 0:
+            res = res[0]
+        print(f"Q: {content}, \nA: {res}")
+        return [res]
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/requirements.txt b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/requirements.txt
new file mode 100644
index 0000000000000000000000000000000000000000..5eb942567d8848695989821165de9381d851e34e
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/requirements.txt
@@ -0,0 +1,49 @@
+accelerate
+aiohttp==3.8.4
+aiosignal==1.3.1
+async-timeout==4.0.2
+attrs==22.2.0
+bitsandbytes==0.37.0
+cchardet==2.1.7
+chardet==5.1.0
+contourpy==1.0.7
+cycler==0.11.0
+filelock==3.9.0
+fonttools==4.38.0
+frozenlist==1.3.3
+huggingface-hub==0.13.4
+importlib-resources==5.12.0
+kiwisolver==1.4.4
+matplotlib==3.7.0
+multidict==6.0.4
+openai==0.27.0
+packaging==23.0
+psutil==5.9.4
+pycocotools==2.0.6
+pyparsing==3.0.9
+python-dateutil==2.8.2
+pyyaml==6.0
+regex==2022.10.31
+tokenizers==0.13.2
+tqdm==4.64.1
+transformers
+timm==0.6.13
+spacy==3.5.1
+webdataset==0.2.48
+scikit-learn==1.2.2
+scipy==1.10.1
+yarl==1.8.2
+zipp==3.14.0
+omegaconf==2.3.0
+opencv-python==4.7.0.72
+iopath==0.1.10
+decord==0.6.0
+tenacity==8.2.2
+peft
+pycocoevalcap
+sentence-transformers
+umap-learn
+notebook
+gradio==3.24.1
+gradio-client==0.0.8
+wandb
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/shell/run_inference.sh b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/shell/run_inference.sh
new file mode 100644
index 0000000000000000000000000000000000000000..f255ff18b50e4734aad8431535962019ad2e842d
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/shell/run_inference.sh
@@ -0,0 +1,15 @@
+export CUDA_VISIBLE_DEVICES="0,1,2,3,4,5,6,7"
+python -m torch.distributed.launch \
+    --nproc_per_node=${NPROC_PER_NODE:-8} \
+    --nnodes=${WORLD_SIZE:-1} \
+    --node_rank=${RANK:-0} \
+    --master_addr=${MASTER_ADDR:-127.0.0.1} \
+    --master_port=${MASTER_PORT:-12345} \
+    ./eval.py \
+    --model_name minicpm \
+    --model_path \
+    --generate_method interleave \
+    --eval_textVQA \
+    --eval_docVQA \
+    --answer_path ./answers \
+    --batchsize 1
\ No newline at end of file
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/shell/run_transform.sh b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/shell/run_transform.sh
new file mode 100644
index 0000000000000000000000000000000000000000..a19a83fba836a366d4893635aee57852b77f2d8b
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/shell/run_transform.sh
@@ -0,0 +1,3 @@
+python ./transform_docvqatest_for_submission.py \
+    --input_file_path \
+    --output_file_path 
\ No newline at end of file
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/transform_docvqatest_for_submission.py b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/transform_docvqatest_for_submission.py
new file mode 100644
index 0000000000000000000000000000000000000000..44048fef9b2c8bc673d449096504fbc708a24762
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/eval_mm/vqaeval/transform_docvqatest_for_submission.py
@@ -0,0 +1,16 @@
+import argparse
+import json
+
+if __name__ == "__main__":
+    parser = argparse.ArgumentParser()
+    parser.add_argument("--input_file_path", type=str, default="", help="path to the originial output json.")
+    parser.add_argument("--output_file_path", type=str, default="", help="path to where you want to save the processed json.")
+    args = parser.parse_args()
+    
+    with open(args.input_file_path , 'r') as f:
+        data = json.load(f)
+
+    transformed_data = [{"questionId": item["question_id"], "answer": item["answer"].replace("</s>", "")} for item in data]
+
+    with open(args.output_file_path, 'w') as f:
+        json.dump(transformed_data, f)
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/finetune/__init__.py b/PyTorch/built-in/mlm/MiniCPM-V/finetune/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/finetune/dataset.py b/PyTorch/built-in/mlm/MiniCPM-V/finetune/dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..92807c38b427eb76b276a4ae0f5de87256383f67
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/finetune/dataset.py
@@ -0,0 +1,458 @@
+import copy
+import json
+import logging
+import math
+import os
+from dataclasses import dataclass, field
+from typing import Dict, List, Optional
+
+import numpy as np
+import torch
+from PIL import Image
+from torch.nn.utils.rnn import pad_sequence
+from torch.utils.data import Dataset
+from transformers import AutoProcessor, AutoTokenizer
+
+llama3_chat_template = "{% set loop_messages = messages %}{% for message in loop_messages %}{% set content = '<|start_header_id|>' + message['role'] + '<|end_header_id|>\n\n'+ message['content'] | trim + '<|eot_id|>' %}{% if loop.index0 == 0 %}{% set content = bos_token + content %}{% endif %}{{ content }}{% endfor %}"
+
+class SupervisedDataset(Dataset):
+    """Dataset for supervised fine-tuning."""
+
+    def __init__(
+        self,
+        raw_data,
+        transform,
+        tokenizer,
+        slice_config,
+        llm_type="minicpm",
+        patch_size=14,
+        query_nums=64,
+        batch_vision=False,
+    ):
+        super(SupervisedDataset, self).__init__()
+        self.raw_data = raw_data
+        self.tokenizer = tokenizer
+        self.transform = transform
+        self.slice_config = slice_config
+        self.llm_type = llm_type
+        self.patch_size = patch_size
+        self.query_nums=query_nums
+        self.batch_vision = batch_vision
+
+    def __len__(self):
+        return len(self.raw_data)
+
+    def __getitem__(self, i) -> Dict[str, torch.Tensor]:
+        image = Image.open(self.raw_data[i]["image"]).convert("RGB")
+        ret = preprocess(
+            image,
+            self.raw_data[i]["conversations"],
+            self.tokenizer,
+            self.transform,
+            query_nums=self.query_nums,
+            slice_config=self.slice_config,
+            llm_type=self.llm_type,
+            patch_size=self.patch_size,
+            batch_vision=self.batch_vision,
+        )
+        ret = dict(
+            input_ids=ret["input_ids"],
+            position_ids=ret["position_ids"],
+            labels=ret["target"],
+            attention_mask=torch.ones_like(ret["input_ids"], dtype=torch.bool),
+            pixel_values=ret["pixel_values"],
+            tgt_sizes=ret["tgt_sizes"],
+            image_bound=ret["image_bound"],
+        )
+
+        return ret
+
+def data_collator(examples, padding_value=0, max_length=2048):
+    def trim_and_pad(seq, batch_first, padding_value):
+        return pad_sequence([s[:max_length] for s in seq], batch_first=True, padding_value=padding_value)
+
+    input_ids = trim_and_pad(
+        [example["input_ids"] for example in examples],
+        batch_first=True,
+        padding_value=padding_value,
+    )
+    position_ids = trim_and_pad(
+        [example["position_ids"] for example in examples],
+        batch_first=True,
+        padding_value=padding_value,
+    )
+    targets = trim_and_pad(
+        [example["labels"] for example in examples],
+        batch_first=True,
+        padding_value=-100,
+    )
+    attention_mask = trim_and_pad(
+        [example["attention_mask"] for example in examples],
+        batch_first=True,
+        padding_value=padding_value,
+    )
+    pixel_values = [example["pixel_values"] for example in examples]
+    image_bound = [example["image_bound"] for example in examples]
+    tgt_sizes = [example["tgt_sizes"] for example in examples]
+    return {
+        "input_ids": input_ids,
+        "position_ids": position_ids,
+        "labels": targets,
+        "attention_mask": attention_mask,
+        "image_bound": image_bound,
+        "tgt_sizes": tgt_sizes,
+        "pixel_values": pixel_values,
+    }
+
+
+def conversation_to_ids(conversation, tokenizer, llm_type=None):
+    """
+    for single image multi-turn conversation
+    conversation: [{'role': 'user', 'content': 'Describe this image'},
+                   {'role': 'assistant', 'content': 'This is a cat.'}]
+    """
+    if llm_type == "llama3":
+        input_ids, context, raw_msg = conversation_to_ids_llama3(
+            conversation, tokenizer
+        )
+    else:
+        input_ids, context, raw_msg = conversation_to_ids_minicpm(
+            conversation, tokenizer
+        )
+
+    ids = torch.from_numpy(np.hstack(input_ids, dtype=np.int32))
+    context = torch.from_numpy(np.hstack(context, dtype=np.int8))
+
+    # build target
+    target = torch.full_like(ids, -100, dtype=torch.int32)
+    for i in range(1, len(ids)):
+        if context[i] == 0:
+            target[i - 1] = ids[i]
+        if context[i] == 1 and context[i - 1] == 0:
+            if hasattr(tokenizer, "eot_id"):
+                target[i - 1] = tokenizer.eot_id
+            else:
+                target[i - 1] = tokenizer.eos_id
+
+    # build image bound
+    image_start_tokens = torch.where(ids == tokenizer.im_start_id)[0]
+    image_start_tokens += 1
+    image_end_tokens = torch.where(ids == tokenizer.im_end_id)[0]
+    if len(image_start_tokens) != len(image_end_tokens):
+        print("image start token != image end tokens")
+        
+    if len(image_start_tokens) > 0:
+        image_bound = torch.hstack(
+            [image_start_tokens.unsqueeze(-1), image_end_tokens.unsqueeze(-1)]
+        )
+    else:
+        image_bound = []
+
+    position_ids = torch.arange(ids.size(0)).long()
+    return {
+        "input_ids": ids,
+        "target": target,
+        "image_bound": image_bound,
+        "raw_msg": raw_msg,
+        "position_ids": position_ids
+    }
+
+
+def conversation_to_ids_minicpm(conversation, tokenizer):
+    raw_msg = ""
+    input_ids = []
+    context = []
+    for idx, msg in enumerate(conversation):
+        role = msg["role"]
+        message = msg["content"]
+        assert role in ["user", "assistant"]
+        if role == "user":
+            prefix = "<用户>"
+        else:
+            prefix = "<AI>"
+        # append eos
+        if idx == len(conversation) - 1:
+            message = message + tokenizer.eos_token
+        prefix_ids = tokenizer.encode(prefix)[1:]  # remove bos
+        message_ids = tokenizer.encode(message)[1:]
+
+        input_ids.append(prefix_ids)
+        input_ids.append(message_ids)
+
+        context.append(np.ones((len(prefix_ids),), dtype=np.int8))
+        if role == "assistant":
+            context.append(np.zeros((len(message_ids),), dtype=np.int8))
+        else:
+            context.append(np.ones((len(message_ids),), dtype=np.int8))
+
+        raw_msg += prefix + message
+
+    return input_ids, context, raw_msg
+
+
+def conversation_to_ids_llama3(conversation, tokenizer):
+    raw_msg = ""
+    input_ids = []
+    context = []
+    raw_msg = tokenizer.apply_chat_template(
+        conversation, tokenize=False, add_generation_prompt=False, chat_template=llama3_chat_template,
+    )
+    input_ids = tokenizer.apply_chat_template(
+        conversation, tokenize=True, add_generation_prompt=False, chat_template=llama3_chat_template,
+    )
+    input_ids = np.array(input_ids)
+
+    start_header_idxs = np.where(
+        input_ids == tokenizer.convert_tokens_to_ids("<|start_header_id|>")
+    )[0]
+    assistant_idxs = np.where(
+        input_ids == tokenizer.convert_tokens_to_ids("assistant")
+    )[0]
+    end_header_idxs = np.where(
+        input_ids == tokenizer.convert_tokens_to_ids("<|end_header_id|>")
+    )[0]
+    eot_idxs = np.where(
+        input_ids == tokenizer.convert_tokens_to_ids("<|eot_id|>"))[0]
+
+    context = np.ones_like(input_ids, dtype=np.int8)
+
+    for assistant_idx in assistant_idxs:
+        if assistant_idx in set((start_header_idxs + end_header_idxs) / 2):
+            st = assistant_idx + 3  # assistant<|end_header_id|>\n\n
+            for eot_idx in eot_idxs:
+                if eot_idx > st:
+                    context[st: eot_idx + 1] = 0
+                    break
+
+    input_ids = np.hstack(input_ids)
+    context = np.hstack(context)
+
+    return input_ids, context, raw_msg
+
+
+def preprocess(
+    image,
+    conversation,
+    tokenizer,
+    transform,
+    query_nums=64,
+    slice_config=None,
+    llm_type=None,
+    patch_size=14,
+    batch_vision=False,
+):
+    """
+    single image preprocess, the image will be placed at the top of the conversation
+    """
+    conversation = copy.deepcopy(conversation)
+    assert len(conversation) > 1, "conversation length must large than 2"
+    assert conversation[0]["role"] == "user", "the first role must be user"
+
+    if slice_config is not None:
+        assert isinstance(slice_config, Dict)
+        assert "patch_size" in slice_config
+        assert "max_slice_nums" in slice_config
+        assert "scale_resolution" in slice_config
+    default_image_placeholder = (
+        tokenizer.im_start + tokenizer.unk_token * query_nums + tokenizer.im_end
+    )
+    if slice_config:
+        images = []
+        source_image, patches, best_grid = slice_image(
+            image,
+            slice_config["max_slice_nums"],
+            slice_config["scale_resolution"],
+            slice_config["patch_size"],
+        )
+        images.append(source_image)
+        image_placeholder = default_image_placeholder
+        if len(patches) > 0:
+            for i in range(len(patches)):
+                for j in range(len(patches[0])):
+                    images.append(patches[i][j])
+
+            image_placeholder += get_grid_placeholder(
+                tokenizer, best_grid, query_nums)
+        images = [transform(i) for i in images]
+    else:
+        images = [transform(image)]
+        image_placeholder = default_image_placeholder
+    if "<image>" in conversation[0]["content"]:
+        conversation[0]["content"] = conversation[0]["content"].replace(
+            "<image>", image_placeholder
+        )
+    else:
+        conversation[0]["content"] = (
+            image_placeholder + "\n" + conversation[0]["content"]
+        )
+
+    input_dict = conversation_to_ids(conversation, tokenizer, llm_type)
+
+    if batch_vision:
+        tgt_sizes = []
+        reshape_images = []
+        for image in images:
+            H, W = image.shape[1:]
+            reshape_image = reshape_by_patch(image, patch_size)
+            reshape_images.append(reshape_image)
+            tgt_sizes.append([H // patch_size, W // patch_size])
+        if tgt_sizes:
+            tgt_sizes = torch.Tensor(tgt_sizes).type(torch.int32)
+
+        input_dict["pixel_values"] = reshape_images
+        input_dict["tgt_sizes"] = tgt_sizes
+
+    else:
+        input_dict["pixel_values"] = images
+        input_dict["tgt_sizes"] = []
+
+    return input_dict
+
+
+def slice_image(
+    image, max_slice_nums=9, scale_resolution=448, patch_size=14, never_split=False
+):
+    original_size = image.size
+    original_width, original_height = original_size
+    log_ratio = math.log(original_width / original_height)
+    ratio = original_width * original_height / \
+        (scale_resolution * scale_resolution)
+    multiple = min(math.ceil(ratio), max_slice_nums)
+
+    source_image = None
+    best_grid = None
+    patches = []
+
+    if multiple <= 1 or never_split:
+        # dont need to slice, upsample
+        best_size = find_best_resize(
+            original_size, scale_resolution, patch_size, allow_upscale=True
+        )
+        source_image = image.resize(best_size, Image.Resampling.BICUBIC)
+    else:
+        candidate_split_grids_nums = []
+        for i in [multiple - 1, multiple, multiple + 1]:
+            if i == 1 or i > max_slice_nums:
+                continue
+            candidate_split_grids_nums.append(i)
+
+        # source image, down-sampling and ensure divided by patch_size
+        best_resize = find_best_resize(
+            original_size, scale_resolution, patch_size)
+        source_image = image.copy().resize(best_resize, Image.Resampling.BICUBIC)
+        candidate_grids = []
+
+        # find best grid
+        for split_grids_nums in candidate_split_grids_nums:
+            m = 1
+            while m <= split_grids_nums:
+                if split_grids_nums % m == 0:
+                    candidate_grids.append([m, split_grids_nums // m])
+                m += 1
+
+        best_grid = [1, 1]
+        min_error = float("inf")
+        for grid in candidate_grids:
+            error = abs(log_ratio - math.log(grid[0] / grid[1]))
+            if error < min_error:
+                best_grid = grid
+                min_error = error
+
+        refine_size = get_refine_size(
+            original_size, best_grid, scale_resolution, patch_size, allow_upscale=True
+        )
+
+        refine_image = image.resize(refine_size, Image.Resampling.BICUBIC)
+        patches = split_to_patches(refine_image, best_grid)
+
+    return source_image, patches, best_grid
+
+
+def ensure_divide(length, patch_size):
+    return max(round(length / patch_size) * patch_size, patch_size)
+
+
+def find_best_resize(original_size, scale_resolution, patch_size, allow_upscale=False):
+    width, height = original_size
+    if (width * height > scale_resolution * scale_resolution) or allow_upscale:
+        r = width / height
+        height = int(scale_resolution / math.sqrt(r))
+        width = int(height * r)
+    best_width = ensure_divide(width, patch_size)
+    best_height = ensure_divide(height, patch_size)
+    return (best_width, best_height)
+
+
+def get_refine_size(
+    original_size, grid, scale_resolution, patch_size, allow_upscale=False
+):
+    width, height = original_size
+    grid_x, grid_y = grid
+
+    refine_width = ensure_divide(width, grid_x)
+    refine_height = ensure_divide(height, grid_y)
+
+    grid_width = refine_width / grid_x
+    grid_height = refine_height / grid_y
+
+    best_grid_size = find_best_resize(
+        (grid_width, grid_height),
+        scale_resolution,
+        patch_size,
+        allow_upscale=allow_upscale,
+    )
+
+    refine_size = (best_grid_size[0] * grid_x, best_grid_size[1] * grid_y)
+
+    return refine_size
+
+
+def split_to_patches(image, grid):
+    patches = []
+    width, height = image.size
+    grid_x = int(width / grid[0])
+    grid_y = int(height / grid[1])
+
+    for i in range(0, height, grid_y):
+        images = []
+        for j in range(0, width, grid_x):
+            box = (j, i, j + grid_x, i + grid_y)
+            patch = image.crop(box)
+            images.append(patch)
+        patches.append(images)
+
+    return patches
+
+
+def get_grid_placeholder(tokenizer, grid, query_num):
+    image_placeholder = (
+        tokenizer.im_start + tokenizer.unk_token * query_num + tokenizer.im_end
+    )
+
+    cols = grid[0]
+    rows = grid[1]
+    slices = []
+    for i in range(rows):
+        lines = []
+        for j in range(cols):
+            lines.append(image_placeholder)
+        slices.append("".join(lines))
+    slice_placeholder = tokenizer.slice_start + \
+        "\n".join(slices) + tokenizer.slice_end
+    return slice_placeholder
+
+
+def reshape_by_patch(image_tensor, patch_size):
+    """
+    :param image_tensor: shape [3, H, W]
+    :param patch_size:
+    :return: [3, patch_size, HW/patch_size]
+    """
+    patches = torch.nn.functional.unfold(
+        image_tensor, (patch_size, patch_size), stride=(patch_size, patch_size)
+    )
+
+    patches = patches.reshape(image_tensor.size(0), patch_size, patch_size, -1)
+    patches = patches.permute(0, 1, 3, 2).reshape(
+        image_tensor.size(0), patch_size, -1)
+    return patches
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/finetune/ds_config_zero2.json b/PyTorch/built-in/mlm/MiniCPM-V/finetune/ds_config_zero2.json
new file mode 100644
index 0000000000000000000000000000000000000000..4d42d440b4b2b55dba2e54863109773b1ee3fee9
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/finetune/ds_config_zero2.json
@@ -0,0 +1,54 @@
+{
+    "fp16": {
+        "enabled": "auto",
+        "loss_scale": 0,
+        "loss_scale_window": 1000,
+        "initial_scale_power": 16,
+        "hysteresis": 2,
+        "min_loss_scale": 1
+    },
+
+    "bf16": {
+        "enabled": "auto"
+    },
+
+    "optimizer": {
+        "type": "AdamW",
+        "params": {
+            "lr": "auto",
+            "betas": "auto",
+            "eps": "auto",
+            "weight_decay": "auto"
+        }
+    },
+
+    "scheduler": {
+        "type": "WarmupLR",
+        "params": {
+            "warmup_min_lr": "auto",
+            "warmup_max_lr": "auto",
+            "warmup_num_steps": "auto"
+        }
+    },
+
+    "zero_optimization": {
+        "stage": 2,
+        "offload_optimizer": {
+            "device": "none",
+            "pin_memory": true
+        },
+        "allgather_partitions": true,
+        "allgather_bucket_size": 2e8,
+        "overlap_comm": true,
+        "reduce_scatter": true,
+        "reduce_bucket_size": 2e8,
+        "contiguous_gradients": true
+    },
+
+    "gradient_accumulation_steps": "auto",
+    "gradient_clipping": "auto",
+    "steps_per_print": 100,
+    "train_batch_size": "auto",
+    "train_micro_batch_size_per_gpu": "auto",
+    "wall_clock_breakdown": false
+}
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/finetune/ds_config_zero3.json b/PyTorch/built-in/mlm/MiniCPM-V/finetune/ds_config_zero3.json
new file mode 100644
index 0000000000000000000000000000000000000000..5c3cd9c9274dee4f57449196edbc07920b3b94ec
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/finetune/ds_config_zero3.json
@@ -0,0 +1,61 @@
+
+{
+    "fp16": {
+        "enabled": "auto",
+        "loss_scale": 0,
+        "loss_scale_window": 1000,
+        "initial_scale_power": 16,
+        "hysteresis": 2,
+        "min_loss_scale": 1
+    },
+    "bf16": {
+        "enabled": "auto"
+    },
+    "optimizer": {
+        "type": "AdamW",
+        "params": {
+            "lr": "auto",
+            "betas": "auto",
+            "eps": "auto",
+            "weight_decay": "auto"
+        }
+    },
+
+    "scheduler": {
+        "type": "WarmupLR",
+        "params": {
+            "warmup_min_lr": "auto",
+            "warmup_max_lr": "auto",
+            "warmup_num_steps": "auto"
+        }
+    },
+
+    "zero_optimization": {
+        "stage": 3,
+        "offload_optimizer": {
+            "device": "none",
+            "pin_memory": true
+        },
+        "offload_param": {
+            "device": "none",
+            "pin_memory": true
+        },
+        "overlap_comm": true,
+        "contiguous_gradients": true,
+        "sub_group_size": 1e9,
+        "reduce_bucket_size": "auto",
+        "stage3_prefetch_bucket_size": "auto",
+        "stage3_param_persistence_threshold": "auto",
+        "stage3_max_live_parameters": 1e9,
+        "stage3_max_reuse_distance": 1e9,
+        "stage3_gather_16bit_weights_on_model_save": true
+    },
+
+    "gradient_accumulation_steps": "auto",
+    "gradient_clipping": "auto",
+    "steps_per_print": 100,
+    "train_batch_size": "auto",
+    "train_micro_batch_size_per_gpu": "auto",
+    "wall_clock_breakdown": false
+}
+
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/finetune/finetune.py b/PyTorch/built-in/mlm/MiniCPM-V/finetune/finetune.py
new file mode 100644
index 0000000000000000000000000000000000000000..44760495a0d66a19faf573b73ad3115924fd2ada
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/finetune/finetune.py
@@ -0,0 +1,282 @@
+import glob
+import json
+import logging
+import os
+from dataclasses import dataclass, field
+from functools import partial
+from typing import Dict, List, Optional, Union, Literal, Tuple
+from types import MethodType
+import torch
+import transformers
+from accelerate.utils import DistributedType
+from deepspeed import zero
+from deepspeed.runtime.zero.partition_parameters import ZeroParamStatus
+
+from transformers import AutoModel, AutoTokenizer
+from transformers.integrations import deepspeed
+from transformers import AutoModel, AutoTokenizer
+
+from dataset import SupervisedDataset, data_collator
+from trainer import CPMTrainer
+
+from peft import LoraConfig, get_peft_model, prepare_model_for_kbit_training
+
+@dataclass
+class ModelArguments:
+    model_name_or_path: Optional[str] = field(default="openbmb/MiniCPM-V-2")
+
+
+@dataclass
+class DataArguments:
+    data_path: str = field(
+        default=None, metadata={"help": "Path to the training data."}
+    )
+    eval_data_path: str = field(
+        default=None, metadata={"help": "Path to the evaluation data."}
+    )
+
+
+@dataclass
+class TrainingArguments(transformers.TrainingArguments):
+    cache_dir: Optional[str] = field(default=None)
+    optim: str = field(default="adamw_torch")
+    model_max_length: int = field(
+        default=2048,
+        metadata={
+            "help": "Maximum sequence length. Sequences will be right padded (and possibly truncated)."
+        },
+    )
+    tune_vision: Optional[bool] = field(default=True)
+    tune_llm: Optional[bool] = field(default=True)
+    llm_type: str = field(default="minicpm")
+    use_lora: Optional[bool] = field(default=False)
+    max_slice_nums: Optional[int] = field(default=9)
+
+
+@dataclass
+class LoraArguments:
+    lora_r: int = 64
+    lora_alpha: int = 64
+    lora_dropout: float = 0.05
+    lora_target_modules: str = r"llm\..*layers\.\d+\.self_attn\.(q_proj|k_proj|v_proj)"
+    lora_weight_path: str = ""
+    lora_bias: str = "none"
+    q_lora: bool = False
+    lora_modules_to_save: str = ""
+    lora_layer_replication: Optional[List[Tuple[int, int]]] = None
+    lora_layers_to_transform: Optional[List[int]] = None
+    lora_layers_pattern: Optional[str] = None
+
+
+local_rank = None
+def rank0_print(*args):
+    if local_rank == 0:
+        print(*args)
+
+
+def safe_save_model_for_hf_trainer(trainer, output_dir: str, bias="none"):
+    """Collects the state dict and dump to disk."""
+    if trainer.args.should_save and trainer.args.local_rank == 0:
+        trainer.save_model(output_dir, )
+
+
+def make_supervised_data_module(
+    tokenizer: transformers.PreTrainedTokenizer,
+    data_args,
+    transform,
+    data_collator=None,
+    llm_type="minicpm",
+    slice_config=None,
+    patch_size=14,
+    query_nums=64,
+    batch_vision=False,
+    max_length=2048,
+) -> Dict:
+    """Make dataset and collator for supervised fine-tuning."""
+    dataset_cls = SupervisedDataset
+
+    rank0_print("Loading data...")
+
+    train_json = json.load(open(data_args.data_path, "r"))
+    train_dataset = dataset_cls(
+        train_json,
+        transform,
+        tokenizer,
+        slice_config=slice_config,
+        llm_type=llm_type,
+        patch_size=patch_size,
+        query_nums=query_nums,
+        batch_vision=batch_vision,
+    )
+
+    if data_args.eval_data_path:
+        eval_json = json.load(open(data_args.eval_data_path, "r"))
+        eval_dataset = dataset_cls(
+            eval_json,
+            transform,
+            tokenizer,
+            slice_config=slice_config,
+            llm_type=llm_type,
+            patch_size=patch_size,
+            query_nums=query_nums,
+            batch_vision=batch_vision,
+        )
+    else:
+        eval_dataset = None
+
+    return dict(
+        train_dataset=train_dataset,
+        eval_dataset=eval_dataset,
+        data_collator= partial(data_collator, max_length=max_length),
+    )
+
+
+def get_parameter_number(model):
+    trainable_params, all_param = 0, 0
+    for param in model.parameters():
+        num_params = param.numel()
+        # if using DS Zero 3 and the weights are initialized empty
+        if num_params == 0 and hasattr(param, "ds_numel"):
+            num_params = param.ds_numel
+
+        all_param += num_params
+        if param.requires_grad:
+            trainable_params += num_params
+
+    return {'Total': all_param, 'Trainable': trainable_params}
+
+
+local_rank = 0
+
+
+def train():
+    seed_all(is_gpu=False, mode=False)
+    global local_rank
+    parser = transformers.HfArgumentParser(
+        (ModelArguments, DataArguments, TrainingArguments, LoraArguments)
+    )
+
+    (
+        model_args,
+        data_args,
+        training_args,
+        lora_args,
+    ) = parser.parse_args_into_dataclasses()
+
+    if getattr(training_args, "deepspeed", None):
+        training_args.distributed_state.distributed_type = DistributedType.DEEPSPEED
+
+    compute_dtype = (
+        torch.float16
+        if training_args.fp16
+        else (torch.bfloat16 if training_args.bf16 else torch.float32)
+    )
+
+    local_rank = training_args.local_rank
+    world_size = int(os.environ.get("WORLD_SIZE", 1))
+    ddp = world_size != 1
+    device_map = None
+    if lora_args.q_lora:
+        device_map = {"": int(os.environ.get("LOCAL_RANK") or 0)} if ddp else None
+        if len(training_args.fsdp) > 0 or deepspeed.is_deepspeed_zero3_enabled():
+            logging.warning(
+                "FSDP or ZeRO3 are not incompatible with QLoRA."
+            )
+
+    model = AutoModel.from_pretrained(
+        model_args.model_name_or_path,
+        trust_remote_code=True,
+        torch_dtype=compute_dtype,
+        device_map=device_map,
+    )
+
+    tokenizer = AutoTokenizer.from_pretrained(
+        model_args.model_name_or_path, trust_remote_code=True
+    )
+
+    if not training_args.tune_vision:
+        model.vpm.requires_grad_(False)
+    if not training_args.tune_llm:
+        model.llm.requires_grad_(False)
+
+    if training_args.use_lora:
+        if training_args.use_lora and training_args.tune_llm:
+            raise ValueError("The model cannot simultaneously adjust LLM parameters and apply LoRA.")
+
+        rank0_print("Currently using LoRA for fine-tuning the MiniCPM-V model.")
+        for name, param in model.llm.named_parameters():
+            param.requires_grad = False
+        modules_to_save = ['embed_tokens', 'resampler']
+        if training_args.tune_vision:
+            modules_to_save.append('vpm')
+        lora_config = LoraConfig(
+            r=lora_args.lora_r,
+            lora_alpha=lora_args.lora_alpha,
+            target_modules=lora_args.lora_target_modules,
+            lora_dropout=lora_args.lora_dropout,
+            bias=lora_args.lora_bias,
+            layers_to_transform=lora_args.lora_layers_to_transform,
+            modules_to_save=modules_to_save,
+        )
+        if not hasattr(model, 'get_input_embeddings'):
+            def get_input_embeddings(self):
+                return self.llm.get_input_embeddings()
+            model.get_input_embeddings = MethodType(get_input_embeddings, model)
+        if lora_args.q_lora:
+            model = prepare_model_for_kbit_training(
+                model, use_gradient_checkpointing=training_args.gradient_checkpointing
+            )
+        model = get_peft_model(model, lora_config)
+        if training_args.gradient_checkpointing:
+            model.enable_input_require_grads()
+
+    rank0_print(get_parameter_number(model))
+
+    llm_type = training_args.llm_type
+
+    rank0_print(f'llm_type={llm_type}')
+
+    # Load data
+    if hasattr(model.config, "slice_config"):
+        model.config.slice_config.max_slice_nums = training_args.max_slice_nums
+        slice_config = model.config.slice_config.to_dict()
+    else:
+        model.config.max_slice_nums = training_args.max_slice_nums
+        slice_config = model.config.to_dict()
+
+    if hasattr(model.config, "batch_vision_input"):
+        batch_vision = model.config.batch_vision_input
+    else:
+        batch_vision = False
+
+    data_module = make_supervised_data_module(
+        tokenizer=tokenizer,
+        data_args=data_args,
+        transform=model.transform,
+        data_collator=data_collator,
+        slice_config=slice_config,
+        llm_type=llm_type,
+        patch_size=model.config.patch_size,
+        query_nums=model.config.query_num,
+        batch_vision=batch_vision,
+        max_length=training_args.model_max_length,
+    )
+
+    trainer = CPMTrainer(
+        model=model,
+        tokenizer=tokenizer,
+        args=training_args,
+        **data_module,
+    )
+
+    trainer.train()
+    trainer.save_state()
+
+    safe_save_model_for_hf_trainer(
+        trainer=trainer,
+        output_dir=training_args.output_dir,
+        bias=lora_args.lora_bias)
+
+
+if __name__ == "__main__":
+    train()
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/finetune/finetune_ds.sh b/PyTorch/built-in/mlm/MiniCPM-V/finetune/finetune_ds.sh
new file mode 100644
index 0000000000000000000000000000000000000000..5dc3a3e67e37b518c1e546e8b01d318862ae04c5
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/finetune/finetune_ds.sh
@@ -0,0 +1,61 @@
+#!/bin/bash
+
+GPUS_PER_NODE=8
+NNODES=1
+NODE_RANK=0
+MASTER_ADDR=localhost
+MASTER_PORT=6001
+
+MODEL="openbmb/MiniCPM-Llama3-V-2_5" # or openbmb/MiniCPM-V-2
+# ATTENTION: specify the path to your training data, which should be a json file consisting of a list of conversations.
+# See the section for finetuning in README for more information.
+DATA="path/to/trainging_data"
+EVAL_DATA="path/to/test_data"
+LLM_TYPE="llama3" # if use openbmb/MiniCPM-V-2, please set LLM_TYPE=minicpm
+
+DISTRIBUTED_ARGS="
+    --nproc_per_node $GPUS_PER_NODE \
+    --nnodes $NNODES \
+    --node_rank $NODE_RANK \
+    --master_addr $MASTER_ADDR \
+    --master_port $MASTER_PORT
+"
+torchrun $DISTRIBUTED_ARGS finetune.py  \
+    --model_name_or_path $MODEL \
+    --llm_type $LLM_TYPE \
+    --data_path $DATA \
+    --eval_data_path $EVAL_DATA \
+    --remove_unused_columns false \
+    --label_names "labels" \
+    --prediction_loss_only false \
+    --bf16 false \
+    --bf16_full_eval false \
+    --fp16 true \
+    --fp16_full_eval true \
+    --do_train \
+    --do_eval \
+    --tune_vision true \
+    --tune_llm true \
+    --model_max_length 2048 \
+    --max_slice_nums 9 \
+    --max_steps 10000 \
+    --eval_steps 1000 \
+    --output_dir output/output_minicpmv2 \
+    --logging_dir output/output_minicpmv2 \
+    --logging_strategy "steps" \
+    --per_device_train_batch_size 1 \
+    --per_device_eval_batch_size 1 \
+    --gradient_accumulation_steps 1 \
+    --evaluation_strategy "steps" \
+    --save_strategy "steps" \
+    --save_steps 1000 \
+    --save_total_limit 10 \
+    --learning_rate 1e-6 \
+    --weight_decay 0.1 \
+    --adam_beta2 0.95 \
+    --warmup_ratio 0.01 \
+    --lr_scheduler_type "cosine" \
+    --logging_steps 1 \
+    --gradient_checkpointing true \
+    --deepspeed ds_config_zero2.json \
+    --report_to "tensorboard" 
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/finetune/finetune_lora.sh b/PyTorch/built-in/mlm/MiniCPM-V/finetune/finetune_lora.sh
new file mode 100644
index 0000000000000000000000000000000000000000..22cf5a290243649a051e3b7a5673067e3067f16b
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/finetune/finetune_lora.sh
@@ -0,0 +1,63 @@
+#!/bin/bash
+
+GPUS_PER_NODE=8
+NNODES=1
+NODE_RANK=0
+MASTER_ADDR=localhost
+MASTER_PORT=6001
+
+MODEL="openbmb/MiniCPM-Llama3-V-2_5" # or openbmb/MiniCPM-V-2
+# ATTENTION: specify the path to your training data, which should be a json file consisting of a list of conversations.
+# See the section for finetuning in README for more information.
+DATA="path/to/trainging_data"
+EVAL_DATA="path/to/test_data"
+LLM_TYPE="llama3" # if use openbmb/MiniCPM-V-2, please set LLM_TYPE=minicpm
+
+DISTRIBUTED_ARGS="
+    --nproc_per_node $GPUS_PER_NODE \
+    --nnodes $NNODES \
+    --node_rank $NODE_RANK \
+    --master_addr $MASTER_ADDR \
+    --master_port $MASTER_PORT
+"
+torchrun $DISTRIBUTED_ARGS finetune.py  \
+    --model_name_or_path $MODEL \
+    --llm_type $LLM_TYPE \
+    --data_path $DATA \
+    --eval_data_path $EVAL_DATA \
+    --remove_unused_columns false \
+    --label_names "labels" \
+    --prediction_loss_only false \
+    --bf16 false \
+    --bf16_full_eval false \
+    --fp16 true \
+    --fp16_full_eval true \
+    --do_train \
+    --do_eval \
+    --tune_vision true \
+    --tune_llm false \
+    --use_lora true \
+    --lora_target_modules "llm\..*layers\.\d+\.self_attn\.(q_proj|k_proj|v_proj|o_proj)" \
+    --model_max_length 2048 \
+    --max_slice_nums 9 \
+    --max_steps 10000 \
+    --eval_steps 1000 \
+    --output_dir output/output_minicpmv2_lora \
+    --logging_dir output/output_minicpmv2_lora \
+    --logging_strategy "steps" \
+    --per_device_train_batch_size 2 \
+    --per_device_eval_batch_size 1 \
+    --gradient_accumulation_steps 1 \
+    --evaluation_strategy "steps" \
+    --save_strategy "steps" \
+    --save_steps 1000 \
+    --save_total_limit 10 \
+    --learning_rate 1e-6 \
+    --weight_decay 0.1 \
+    --adam_beta2 0.95 \
+    --warmup_ratio 0.01 \
+    --lr_scheduler_type "cosine" \
+    --logging_steps 1 \
+    --gradient_checkpointing true \
+    --deepspeed ds_config_zero2.json \
+    --report_to "tensorboard" # wandb
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/finetune/readme.md b/PyTorch/built-in/mlm/MiniCPM-V/finetune/readme.md
new file mode 100644
index 0000000000000000000000000000000000000000..2f4e095ccce66c7f26d2e5ad71a1b8dd45421acc
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/finetune/readme.md
@@ -0,0 +1,232 @@
+# MiniCPM-V Finetuning
+
+
+We offer the official scripts for easy finetuning of the pretrained **MiniCPM-Llama3-V 2.5** and **MiniCPM-V 2.0** on downstream tasks. Our finetune scripts use transformers Trainer and DeepSpeed by default.
+
+### Data preparation
+
+To prepare your finetuning data, you should formulate each sample as a dictionary consisting of an id, an image path list with an image, and a list of conversations. Then save data samples in JSON files.
+
+For the vision-language example with image, you are required to provide **\<image\>** to define the position to insert the image embeddings. If you don't provide \<image\>, the image will be placed at the front of the conversation.
+
+<details>
+  <summary>
+    <b>vision-language example (vl_finetune_data.json) with 1 samples.</b>
+  </summary>
+
+```
+  [
+    {
+      "id": "0",
+      "image": 'path/to/image_0.jpg',
+      "conversations": [
+            {
+              'role': 'user', 
+              'content': '<image>\nHow many desserts are on the white plate?'
+            }, 
+            {
+                'role': 'assistant', 
+                'content': 'There are three desserts on the white plate.'
+            },   
+            {
+                'role': 'user', 
+                'content': 'What type of desserts are they?'
+            },
+            {
+                'role': 'assistant', 
+                'content': 'The desserts are cakes with bananas and pecans on top. They share similarities with donuts, but the presence of bananas and pecans differentiates them.'
+            }, 
+            {
+                'role': 'user', 
+                'content': 'What is the setting of the image?'}, 
+            {
+                'role': 'assistant', 
+                'content': 'The image is set on a table top with a plate containing the three desserts.'
+            },
+        ]
+    },
+  ]
+```
+
+</details>
+
+### Full-parameter finetuning
+
+Full-parameter parameter finetuning requires updating all parameters of LLM in the whole training process. Please specify the correct MODEL path, DATA path and LLM_TYPE in the shell scripts.
+
+```shell
+MODEL="openbmb/MiniCPM-Llama3-V-2_5" # or openbmb/MiniCPM-V-2
+DATA="path/to/trainging_data" # json file
+EVAL_DATA="path/to/test_data" # json file
+LLM_TYPE="llama3" # if use openbmb/MiniCPM-V-2, please set LLM_TYPE=minicpm
+```
+
+To launch your training, run the following script:
+
+```
+sh finetune_ds.sh
+```
+
+Specially, Llama3 has a different chat_template for training and inference, we modified the chat_template for training, so please take care to restore the chat_template when inference on the training ckpt.
+
+### LoRA finetuning
+
+The LoRA allows light-weight model tuning with only a small subset of parameters updated. We provide the LoRA implementation based on `peft`. To launch your training, run the following script:
+
+```
+sh finetune_lora.sh
+```
+
+After training, you could load the model with the path to the adapter. We advise you to use absolute path for your pretrained model. This is because LoRA only saves the adapter and the absolute path in the adapter configuration json file is used for finding out the pretrained model to load.
+
+```
+from peft import AutoPeftModelForCausalLM
+
+path_to_adapter="path_to_adapter"
+
+model = AutoPeftModelForCausalLM.from_pretrained(
+    # path to the output directory
+    path_to_adapter,
+    device_map="auto",
+    trust_remote_code=True
+).eval()
+
+vpm_resampler_embedtokens_weight = torch.load(f"{path_to_adapter}/vpm_resampler_embedtokens.pt")
+
+msg = model.load_state_dict(vpm_resampler_embedtokens_weight, strict=False)
+```
+
+
+### Model Fine-tuning Memory Usage Statistics
+
+The following table presents the memory usage of the model when fine-tuning using NVIDIA A100 (80GiB) GPUs under different numbers of GPUs. The fine-tuning was performed with the DeepSpeed Zero-3 optimization, Gradient Checkpointing techniques and offloading optimizer as well as parameters memory to cpu, with a maximum length set to 2048 and batch size set to 1. You refer to [deepspeed zero stage](https://huggingface.co/docs/transformers/v4.41.2/en/deepspeed#select-a-zero-stage) to reduce memory cost.
+
+| Fine-tuning Method | GPUs: 2 | GPUs: 4 | GPUs: 8 |
+|--------------------|---------|---------|---------|
+| LoRA Fine-tuning   | 14.4 GiB| 13.6 GiB|   13.1 GiB   |
+| Full Parameters Fine-tuning | 16.0 GiB | 15.8 GiB | 15.63GiB |
+
+### Notes
+- **Fine-tuning Method**: Displays two different fine-tuning strategies, LoRA fine-tuning and Full parameters fine-tuning.
+- **Number of GPUs**: The table lists the memory usage for configurations with 2, 4, and 8 GPUs.
+- **Memory Usage**: Expressed in GiB, this shows the required memory for each fine-tuning method under corresponding GPU configurations.
+- **Out of memory**: Indicates that the memory was insufficient for full parameters fine-tuning under the current GPU configurations.
+
+### Finetuning FAQs
+
+<details>
+<summary>Q:When you encounter Out of Memory (OOM) issues during training large models, you can try the following methods to resolve or mitigate the issue:</summary>
+
+A：When you face Out of Memory (OOM) issues during training large models, the following strategies may help resolve or mitigate the problem:
+#### Adjust Model Hyperparameters
+- **Reduce `max_model_length`**: Decreasing the maximum sequence length the model processes can significantly reduce the memory required for each operation. For example, reducing the maximum length from 2048 to 1200 or another value suitable for your dataset.
+```
+--model_max_length 1200
+
+```
+- **Lower `batch_size`**: Reducing the amount of data processed in each batch helps decrease memory consumption.
+```
+--batch_size 1
+ ```
+- **Reduce the number of slices (`slice`)**: When handling large datasets such as large images files, reducing the number of slices processed each time can lower memory requirements.
+```
+--max_slice_nums 9 
+```
+
+#### Reduce Training Model Parameters
+- **Do not train VPM (Visual Processing Module)**: You can adjust hyperparameters in the finetune script to opt out of training the visual processing module to save memory.
+```
+--tune_vision false
+```
+- **Use LoRA finetuning**: Refer to the [LoRA finetuning](#LoRA-finetuning) section.
+
+#### Optimize with DeepSpeed
+- **Configure DeepSpeed Zero Stage 2**: Use the following configuration to offload optimizer parameters to the CPU, reducing memory pressure on the GPU:
+  ```json
+  "zero_optimization": {
+    "stage": 2,
+    "offload_optimizer": {
+      "device": "cpu",
+      "pin_memory": true
+    }
+  }
+- **Configure DeepSpeed Zero Stage 3**：Further offload model parameters and optimizer parameters to the CPU, further reducing GPU memory usage:
+```json
+"zero_optimization": {
+  "stage": 3,
+  "offload_optimizer": {
+    "device": "cpu",
+    "pin_memory": true
+  },
+  "offload_param": {
+    "device": "cpu",
+    "pin_memory": true
+  }
+}
+```
+You can visit [huggingface deepspeed](https://huggingface.co/docs/transformers/deepspeed) to find out more about how to use DeepSpeed.
+</details>
+<details>
+<summary>Q: Encounter an error while using the AutoPeftModelForCausalLM to load a checkpoint that has undergone lora fine-tuning</summary>
+
+A: The error as described in [issues 168](https://github.com/OpenBMB/MiniCPM-V/issues/168) occurs because the model lacks `get_input_embeddings` and `set_input_embeddings` methods. Follow these steps to resolve this issue: 
+
+1.**Reload the Fine-Tuned Model:** Make sure you correctly load the checkpoint that has been fine-tuned using lora techniques. Use the following code example to guide you:
+   ```python
+   from peft import AutoPeftModelForCausalLM
+
+   model = AutoPeftModelForCausalLM.from_pretrained(
+       'path_to_your_fine_tuned_checkpoint',  # Path to your fine-tuned checkpoint directory
+       output='output/minicpmv2_lora',
+       device_map='auto',
+       trust_remote_code=True
+   ).eval()
+   ```
+  2.**Update the `model_minicpmv.py` File:**
+   - **Verification:** Make sure you verify and update your `model_minicpmv.py` file to ensure it is the latest version.
+   - **Update Hugging Face Library Code:** If the issue persists after updating the file, consider updating the related code in the Hugging Face library.
+   - **Direct File Copy:** For a quick resolution, directly download and copy the latest `model_minicpmv.py` file into your project. This file is available from the following sources:
+     - [MiniCPM-Llama3-V-2_5 on Hugging Face](https://huggingface.co/openbmb/MiniCPM-Llama3-V-2_5/tree/main)
+     - [MiniCPM-V-2 on Hugging Face](https://huggingface.co/openbmb/MiniCPM-V-2)
+</details>
+
+<details>
+<summary>Q: How do I use the `flash_attention_2` implementation when loading a pretrained model?</summary>
+
+A: If your environment supports `flash_attn2`, you can add an argument `_attn_implementation="flash_attention_2"` when using the `AutoModel.from_pretrained` method to load a model. For example:
+
+```python
+model = AutoModel.from_pretrained('model_name', _attn_implementation="flash_attention_2")
+```
+</details>
+
+<details>
+<summary>Q: What if our data is resized to 512? Can we use the original image size instead?</summary>
+
+A: Our model supports up to 1344x1344 lossless encoding. If you are currently resizing your images to 512, you might want to try using the original image sizes instead. Our system automatically includes a high-definition image encoding scheme by default.
+
+</details>
+
+<details>
+<summary>Q: What should we do if we encounter out-of-memory (OOM) errors?</summary>
+
+A: If you experience OOM issues, consider reducing the batch size (`bs`). To maintain an equivalent total batch size, you can adjust the `gradient_accumulation_steps` setting. This approach allows you to manage memory usage effectively while still processing the desired amount of data per training step.
+</details>
+
+<details>
+<summary>Q: How can we determine the maximum length for our training data, and what if we do not want to train the vision encoder?</summary>
+
+A: I recommend using this function [here](https://github.com/OpenBMB/MiniCPM-V/blob/main/finetune/dataset.py#L220) to sample the length of your training data. Note that the `input_ids` length includes the image portion. Once you determine the maximum length, you can specify it in the startup command using `--model_max_length xxx`.
+
+Additionally, if you prefer not to train the vision encoder, you can add `--tune_vision false` to your command.
+
+</details>
+
+<details>
+<summary>Q: How can we adjust training hyperparameters when using LoRA to train our model?</summary>
+
+A: You can refer to the [LoRA documentation](https://huggingface.co/docs/peft/en/package_reference/lora#peft.LoraConfig) for guidance on adjusting your training hyperparameters when using LoRA. This documentation provides detailed information on configuring various parameters specific to the LoRA adaptation technique.
+</details>
+
+#### Customizing Hyperparameters
+To tailor the training process according to your specific requirements, you can adjust various hyperparameters. For comprehensive documentation on available hyperparameters and their functionalities, you can refer to the [official Transformers documentation](https://huggingface.co/docs/transformers/main_classes/trainer#transformers.TrainingArguments) and [Lora documentation](https://huggingface.co/docs/peft/en/package_reference/lora#peft.LoraConfig). Experimentation and fine-tuning of these parameters are essential for achieving optimal model performance tailored to your specific task and dataset.
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/finetune/trainer.py b/PyTorch/built-in/mlm/MiniCPM-V/finetune/trainer.py
new file mode 100644
index 0000000000000000000000000000000000000000..fa57bd047b16855056e2cc83da7e7fa553be51c3
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/finetune/trainer.py
@@ -0,0 +1,271 @@
+import torch
+import torch.nn as nn
+import deepspeed
+from transformers import Trainer
+from transformers.trainer_pt_utils import nested_detach
+from transformers.utils import is_sagemaker_mp_enabled
+from transformers.trainer import *
+from transformers.integrations import is_deepspeed_zero3_enabled
+
+
+class CPMTrainer(Trainer):
+    def compute_loss(self, model, inputs, return_outputs=False):
+        if "labels" in inputs:
+            labels = inputs.pop("labels")
+        else:
+            labels = None
+        self.model.resampler.pos_embed = self.model.resampler.pos_embed.to(self.model.device)
+        if is_deepspeed_zero3_enabled():
+            with deepspeed.zero.GatheredParameters(self.model.resampler.attn.parameters(), modifier_rank=0):
+                if not self.args.use_lora:
+                    outputs = self.model(data = inputs, use_cache=False)
+                else:
+                    with self.model._enable_peft_forward_hooks(**inputs):
+                        outputs = self.model.base_model(data = inputs, use_cache=False)
+        else:
+            if not self.args.use_lora:
+                outputs = self.model(data = inputs, use_cache=False)
+            else:
+                with self.model._enable_peft_forward_hooks(**inputs):
+                    outputs = self.model.base_model(data = inputs, use_cache=False)
+                
+        if labels is not None:
+            # Flatten the tokens
+            loss_fct = nn.CrossEntropyLoss()
+            logits = outputs.logits.view(-1,
+                                         self.model.config.vocab_size).contiguous()
+            labels = labels.view(-1).long().contiguous()
+            # Enable model parallelism
+            labels = labels.to(logits.device)
+            loss = loss_fct(logits, labels)
+        else:
+            if isinstance(outputs, dict) and "loss" not in outputs:
+                raise ValueError(
+                    "The model did not return a loss from the inputs, only the following keys: "
+                    f"{','.join(outputs.keys())}. For reference, the inputs it received are {','.join(inputs.keys())}."
+                )
+            # We don't use .loss here since the model may return tuples instead of ModelOutput.
+            loss = outputs["loss"] if isinstance(outputs, dict) else outputs[0]
+
+        return (loss, outputs) if return_outputs else loss
+
+    def prediction_step(
+        self,
+        model: nn.Module,
+        inputs: Dict[str, Union[torch.Tensor, Any]],
+        prediction_loss_only: bool,
+        ignore_keys: Optional[List[str]] = None,
+    ) -> Tuple[Optional[torch.Tensor], Optional[torch.Tensor], Optional[torch.Tensor]]:
+        """
+        Perform an evaluation step on `model` using `inputs`.
+
+        Subclass and override to inject custom behavior.
+
+        Args:
+            model (`nn.Module`):
+                The model to evaluate.
+            inputs (`Dict[str, Union[torch.Tensor, Any]]`):
+                The inputs and targets of the model.
+
+                The dictionary will be unpacked before being fed to the model. Most models expect the targets under the
+                argument `labels`. Check your model's documentation for all accepted arguments.
+            prediction_loss_only (`bool`):
+                Whether or not to return the loss only.
+            ignore_keys (`List[str]`, *optional*):
+                A list of keys in the output of your model (if it is a dictionary) that should be ignored when
+                gathering predictions.
+
+        Return:
+            Tuple[Optional[torch.Tensor], Optional[torch.Tensor], Optional[torch.Tensor]]: A tuple with the loss,
+            logits and labels (each being optional).
+        """
+        has_labels = (
+            False
+            if len(self.label_names) == 0
+            else all(inputs.get(k) is not None for k in self.label_names)
+        )
+        # For CLIP-like models capable of returning loss values.
+        # If `return_loss` is not specified or being `None` in `inputs`, we check if the default value of `return_loss`
+        # is `True` in `model.forward`.
+        return_loss = inputs.get("return_loss", None)
+        if return_loss is None:
+            return_loss = self.can_return_loss
+        loss_without_labels = (
+            True if len(self.label_names) == 0 and return_loss else False
+        )
+
+        inputs = self._prepare_inputs(inputs)
+        if ignore_keys is None:
+            if hasattr(self.model, "config"):
+                ignore_keys = getattr(
+                    self.model.config, "keys_to_ignore_at_inference", []
+                )
+            else:
+                ignore_keys = []
+
+        # labels may be popped when computing the loss (label smoothing for instance) so we grab them first.
+        if has_labels or loss_without_labels:
+            labels = nested_detach(tuple(inputs.get(name)
+                                   for name in self.label_names))
+            if len(labels) == 1:
+                labels = labels[0]
+        else:
+            labels = None
+
+        with torch.no_grad():
+            if is_sagemaker_mp_enabled():
+                raw_outputs = smp_forward_only(model, inputs)
+                if has_labels or loss_without_labels:
+                    if isinstance(raw_outputs, dict):
+                        loss_mb = raw_outputs["loss"]
+                        logits_mb = tuple(
+                            v
+                            for k, v in raw_outputs.items()
+                            if k not in ignore_keys + ["loss"]
+                        )
+                    else:
+                        loss_mb = raw_outputs[0]
+                        logits_mb = raw_outputs[1:]
+
+                    loss = loss_mb.reduce_mean().detach().cpu()
+                    logits = smp_nested_concat(logits_mb)
+                else:
+                    loss = None
+                    if isinstance(raw_outputs, dict):
+                        logits_mb = tuple(
+                            v for k, v in raw_outputs.items() if k not in ignore_keys
+                        )
+                    else:
+                        logits_mb = raw_outputs
+                    logits = smp_nested_concat(logits_mb)
+            else:
+                if has_labels or loss_without_labels:
+                    with self.compute_loss_context_manager():
+                        loss, outputs = self.compute_loss(
+                            model, inputs, return_outputs=True
+                        )
+                    loss = loss.mean().detach()
+
+                    if isinstance(outputs, dict):
+                        logits = tuple(
+                            v
+                            for k, v in outputs.items()
+                            if k not in ignore_keys + ["loss"]
+                        )
+                    else:
+                        logits = outputs[1:]
+                else:
+                    loss = None
+                    with self.compute_loss_context_manager():
+                        outputs = model(**inputs)
+                    if isinstance(outputs, dict):
+                        logits = tuple(
+                            v for k, v in outputs.items() if k not in ignore_keys
+                        )
+                    else:
+                        logits = outputs
+                    # TODO: this needs to be fixed and made cleaner later.
+                    if self.args.past_index >= 0:
+                        self._past = outputs[self.args.past_index - 1]
+
+        if prediction_loss_only:
+            return (loss, None, None)
+
+        logits = nested_detach(logits)
+        if len(logits) == 1:
+            logits = logits[0]
+
+        return (loss, logits, labels)
+        
+    def training_step(self, model: nn.Module, inputs: Dict[str, Union[torch.Tensor, Any]]) -> torch.Tensor:
+        """
+        Perform a training step on a batch of inputs.
+
+        Subclass and override to inject custom behavior.
+
+        Args:
+            model (`nn.Module`):
+                The model to train.
+            inputs (`Dict[str, Union[torch.Tensor, Any]]`):
+                The inputs and targets of the model.
+
+                The dictionary will be unpacked before being fed to the model. Most models expect the targets under the
+                argument `labels`. Check your model's documentation for all accepted arguments.
+
+        Return:
+            `torch.Tensor`: The tensor with training loss on this batch.
+        """
+        model.train()
+        inputs = self._prepare_inputs(inputs)
+
+        if is_sagemaker_mp_enabled():
+            loss_mb = smp_forward_backward(model, inputs, self.args.gradient_accumulation_steps)
+            return loss_mb.reduce_mean().detach().to(self.args.device)
+
+        with self.compute_loss_context_manager():
+            loss = self.compute_loss(model, inputs)
+
+        del inputs
+        torch.cuda.empty_cache()
+
+        if self.args.n_gpu > 1:
+            loss = loss.mean()  # mean() to average on multi-gpu parallel training
+
+        if self.use_apex:
+            with amp.scale_loss(loss, self.optimizer) as scaled_loss:
+                scaled_loss.backward()
+        else:
+            if is_deepspeed_zero3_enabled():
+                with deepspeed.zero.GatheredParameters(self.model.resampler.attn.parameters(), modifier_rank=0):
+                    self.accelerator.backward(loss)
+            else:
+                self.accelerator.backward(loss)
+
+        return loss.detach() / self.args.gradient_accumulation_steps
+    
+    def _save(self, output_dir: Optional[str] = None, state_dict=None):
+        # If we are executing this function, we are the process zero, so we don't check for that.
+        output_dir = output_dir if output_dir is not None else self.args.output_dir
+        os.makedirs(output_dir, exist_ok=True)
+        logger.info(f"Saving model checkpoint to {output_dir}")
+
+        supported_classes = (PreTrainedModel,) if not is_peft_available() else (PreTrainedModel, PeftModel)
+        # Save a trained model and configuration using `save_pretrained()`.
+        # They can then be reloaded using `from_pretrained()`
+        if not isinstance(self.model, supported_classes):
+            if state_dict is None:
+                state_dict = self.model.state_dict()
+
+            if isinstance(unwrap_model(self.model), supported_classes):
+                unwrap_model(self.model).save_pretrained(
+                    output_dir, state_dict=state_dict, safe_serialization=self.args.save_safetensors
+                )
+            else:
+                logger.info("Trainer.model is not a `PreTrainedModel`, only saving its state dict.")
+                if self.args.save_safetensors:
+                    safetensors.torch.save_file(
+                        state_dict, os.path.join(output_dir, SAFE_WEIGHTS_NAME), metadata={"format": "pt"}
+                    )
+                else:
+                    torch.save(state_dict, os.path.join(output_dir, WEIGHTS_NAME))
+        else:
+            if self.args.use_lora:
+                from collections import OrderedDict
+                state_dict_vision = OrderedDict()
+                for key, values in state_dict.items():
+                    if 'vpm' in key or 'resampler' in key or 'embed_tokens' in key:
+                        state_dict_vision[key] = values
+                self.model.save_pretrained(
+                    output_dir, state_dict=state_dict, safe_serialization=self.args.save_safetensors
+                )
+                torch.save(state_dict_vision, f"{output_dir}/vpm_resampler_embedtokens.pt", )
+            else:
+                self.model.save_pretrained(
+                    output_dir, state_dict=state_dict, safe_serialization=self.args.save_safetensors
+                )
+
+        if self.tokenizer is not None:
+            self.tokenizer.save_pretrained(output_dir)
+
+        # Good practice: save your training arguments together with the trained model
+        torch.save(self.args, os.path.join(output_dir, TRAINING_ARGS_NAME))
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/huggingface_modify/configuration_minicpm.py b/PyTorch/built-in/mlm/MiniCPM-V/huggingface_modify/configuration_minicpm.py
new file mode 100644
index 0000000000000000000000000000000000000000..ac8a95a909ce75f0939e9d5ad34676f81e5a588f
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/huggingface_modify/configuration_minicpm.py
@@ -0,0 +1,113 @@
+# coding=utf-8
+# Copyright 2022 EleutherAI and the HuggingFace Inc. team. All rights reserved.
+#
+# This code is based on EleutherAI's GPT-NeoX library and the GPT-NeoX
+# and OPT implementations in this library. It has been modified from its
+# original forms to accommodate minor architectural differences compared
+# to GPT-NeoX and OPT used by the Meta AI team that trained the model.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+""" MiniCPM model configuration"""
+import os
+from typing import Union
+
+from transformers.utils import logging
+from transformers import LlamaConfig, PretrainedConfig
+from transformers.models.idefics2.modeling_idefics2 import Idefics2VisionConfig
+
+logger = logging.get_logger(__name__)
+
+
+class MiniCPMVSliceConfig(PretrainedConfig):
+    model_type = "minicpmv"
+
+    def __init__(
+        self,
+        patch_size=14,
+        max_slice_nums=9,
+        scale_resolution=448,
+        **kwargs,
+    ):
+        super().__init__(**kwargs)
+        self.patch_size = patch_size
+        self.max_slice_nums = max_slice_nums
+        self.scale_resolution = scale_resolution
+
+    @classmethod
+    def from_pretrained(cls, pretrained_model_name_or_path: Union[str, os.PathLike], **kwargs) -> "PretrainedConfig":
+        cls._set_token_in_kwargs(kwargs)
+
+        config_dict, kwargs = cls.get_config_dict(pretrained_model_name_or_path, **kwargs)
+
+        if config_dict.get("model_type") == "minicpmv":
+            config_dict = config_dict["slice_config"]
+
+        if "model_type" in config_dict and hasattr(cls, "model_type") and config_dict["model_type"] != cls.model_type:
+            logger.warning(
+                f"You are using a model of type {config_dict['model_type']} to instantiate a model of type "
+                f"{cls.model_type}. This is not supported for all configurations of models and can yield errors."
+            )
+
+        return cls.from_dict(config_dict, **kwargs)
+
+
+
+class MiniCPMVConfig(LlamaConfig):
+    model_type = "minicpmv"
+    keys_to_ignore_at_inference = ["past_key_values"]
+
+    default_vision_config = {
+        "hidden_size": 1152,
+        "image_size": 980,
+        "intermediate_size": 4304,
+        "model_type": "idefics2",
+        "num_attention_heads": 16,
+        "num_hidden_layers": 27,
+        "patch_size": 14,
+    }
+
+    def __init__(
+        self,
+        use_cache=True,
+        query_num=64,
+        image_size=448,
+        drop_vision_last_layer=True,
+        batch_vision_input=True,
+        slice_config=None,
+        vision_config=None,
+        **kwargs,
+    ):
+        self.use_cache = use_cache
+        self.query_num = query_num
+        self.image_size = image_size
+        self.drop_vision_last_layer = drop_vision_last_layer
+        self.batch_vision_input = batch_vision_input
+
+        if slice_config is None:
+            self.slice_config = MiniCPMVSliceConfig(max_slice_nums=1)
+        else:
+            self.slice_config = MiniCPMVSliceConfig(**slice_config)
+        self.slice_mode = True
+
+        # same as HuggingFaceM4/siglip-so400m-14-980-flash-attn2-navit
+        if vision_config is None:
+            self.vision_config = Idefics2VisionConfig(**self.default_vision_config)
+            logger.info("vision_config is None, using default vision config")
+        elif isinstance(vision_config, dict):
+            self.vision_config = Idefics2VisionConfig(**vision_config)
+        elif isinstance(vision_config, Idefics2VisionConfig):
+            self.vision_config = vision_config
+
+        self.patch_size = self.vision_config.patch_size
+
+        super().__init__(**kwargs)
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/huggingface_modify/modeling_minicpmv.py b/PyTorch/built-in/mlm/MiniCPM-V/huggingface_modify/modeling_minicpmv.py
new file mode 100644
index 0000000000000000000000000000000000000000..d0c5186c2c763e0b8531b2424f59ee85200ba3ed
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/huggingface_modify/modeling_minicpmv.py
@@ -0,0 +1,368 @@
+import math
+import json
+import torch
+from threading import Thread
+from copy import deepcopy
+from PIL import Image
+from torchvision import transforms
+from transformers import LlamaPreTrainedModel, LlamaForCausalLM, TextIteratorStreamer
+from transformers.models.idefics2.modeling_idefics2 import Idefics2VisionTransformer
+from transformers import AutoProcessor
+
+from .configuration_minicpm import MiniCPMVConfig
+from .resampler import Resampler
+
+IMAGENET_INCEPTION_MEAN = (0.5, 0.5, 0.5)  # timm.data.IMAGENET_INCEPTION_MEAN
+IMAGENET_INCEPTION_STD = (0.5, 0.5, 0.5)  # timm.data.IMAGENET_INCEPTION_STD
+
+
+class MiniCPMVPreTrainedModel(LlamaPreTrainedModel):
+    config_class = MiniCPMVConfig
+
+
+class MiniCPMV(MiniCPMVPreTrainedModel):
+    def __init__(self, config):
+        super().__init__(config)
+
+        self.llm = LlamaForCausalLM(config)
+        self.vpm = self.init_vision_module()
+        self.vision_dim = self.vpm.embed_dim
+        self.embed_dim = self.llm.config.hidden_size
+        self.resampler = self.init_resampler(self.embed_dim, self.vision_dim)
+        self.transform = self.init_transform()
+
+    def init_vision_module(self):
+        # same as HuggingFaceM4/siglip-so400m-14-980-flash-attn2-navit
+        model = Idefics2VisionTransformer(self.config.vision_config)
+        if self.config.drop_vision_last_layer:
+            model.encoder.layers = model.encoder.layers[:-1]
+
+        setattr(model, 'embed_dim', model.embeddings.embed_dim)
+        setattr(model, 'patch_size', model.embeddings.patch_size)
+
+        return model
+
+    def init_resampler(self, embed_dim, vision_dim):
+        return Resampler(
+            num_queries=self.config.query_num,
+            embed_dim=embed_dim,
+            num_heads=embed_dim // 128,
+            kv_dim=vision_dim,
+            adaptive=True
+        )
+
+    def init_transform(self):
+        return transforms.Compose(
+            [
+                transforms.ToTensor(),
+                transforms.Normalize(
+                    mean=IMAGENET_INCEPTION_MEAN, std=IMAGENET_INCEPTION_STD
+                ),
+            ]
+        )
+
+    def get_input_embeddings(self):
+        return self.llm.get_input_embeddings()
+
+    def set_input_embeddings(self, value):
+        self.llm.embed_tokens = value
+
+    def get_output_embeddings(self):
+        return self.llm.lm_head
+
+    def set_output_embeddings(self, new_embeddings):
+        self.llm.lm_head = new_embeddings
+
+    def set_decoder(self, decoder):
+        self.llm = decoder
+
+    def get_decoder(self):
+        return self.llm
+
+    def get_vllm_embedding(self, data):
+        if 'vision_hidden_states' not in data:
+            dtype = self.llm.model.embed_tokens.weight.dtype
+            device = self.llm.model.embed_tokens.weight.device
+            tgt_sizes = data['tgt_sizes']
+            pixel_values_list = data['pixel_values']
+            vision_hidden_states = []
+            all_pixel_values = []
+            img_cnt = []
+            for pixel_values in pixel_values_list:
+                img_cnt.append(len(pixel_values))
+                all_pixel_values.extend([i.flatten(end_dim=1).permute(1, 0) for i in pixel_values])
+
+            # exist image
+            if all_pixel_values:
+                tgt_sizes = torch.vstack(tgt_sizes).type(torch.int32)
+
+                if self.config.batch_vision_input:
+                    max_patches = torch.max(tgt_sizes[:, 0] * tgt_sizes[:, 1])
+
+                    all_pixel_values = torch.nn.utils.rnn.pad_sequence(all_pixel_values, batch_first=True,
+                                                                       padding_value=0.0)
+                    B, L, _ = all_pixel_values.shape
+                    all_pixel_values = all_pixel_values.permute(0, 2, 1).reshape(B, 3, -1, L)
+
+                    patch_attn_mask = torch.zeros((B, 1, max_patches), dtype=torch.bool, device=device)
+                    for i in range(B):
+                        patch_attn_mask[i, :tgt_sizes[i][0] * tgt_sizes[i][1]] = True
+
+                    vision_embedding = self.vpm(all_pixel_values.type(dtype),
+                                                patch_attention_mask=patch_attn_mask).last_hidden_state
+                    vision_embedding = self.resampler(vision_embedding, tgt_sizes)
+                else:
+                    # get vision_embedding foreach
+                    vision_embedding = []
+                    for single_tgt_size, single_pixel_values in zip(tgt_sizes, all_pixel_values):
+                        single_pixel_values = single_pixel_values.unsqueeze(0)
+                        B, L, _ = single_pixel_values.shape
+                        single_pixel_values = single_pixel_values.permute(0, 2, 1).reshape(B, 3, -1, L)
+                        single_vision_embedding = self.vpm(single_pixel_values.type(dtype)).last_hidden_state
+                        single_vision_embedding = self.resampler(single_vision_embedding, single_tgt_size.unsqueeze(0))
+                        vision_embedding.append(single_vision_embedding)
+                    vision_embedding = torch.vstack(vision_embedding)
+
+                start = 0
+                for pixel_values in pixel_values_list:
+                    img_cnt = len(pixel_values)
+                    if img_cnt > 0:
+                        vision_hidden_states.append(vision_embedding[start: start + img_cnt])
+                        start += img_cnt
+                    else:
+                        vision_hidden_states.append([])
+            else:  # no image
+                if self.training:
+                    dummy_image = torch.zeros(
+                        (1, 3, 224, 224),
+                        device=device, dtype=dtype
+                    )
+                    tgt_sizes = torch.Tensor(
+                        [[(224 // self.config.patch_size), math.ceil(224 / self.config.patch_size)]]).type(torch.int32)
+                    dummy_feature = self.resampler(self.vpm(dummy_image).last_hidden_state, tgt_sizes)
+                else:
+                    dummy_feature = []
+                for _ in range(len(pixel_values_list)):
+                    vision_hidden_states.append(dummy_feature)
+
+        else:
+            vision_hidden_states = data['vision_hidden_states']
+
+        if hasattr(self.llm.config, 'scale_emb'):
+            vllm_embedding = self.llm.model.embed_tokens(data['input_ids']) * self.llm.config.scale_emb
+        else:
+            vllm_embedding = self.llm.model.embed_tokens(data['input_ids'])
+
+        vision_hidden_states = [i.type(vllm_embedding.dtype) if isinstance(
+            i, torch.Tensor) else i for i in vision_hidden_states]
+
+        bs = len(data['input_ids'])
+        for i in range(bs):
+            cur_vs_hs = vision_hidden_states[i]
+            if len(cur_vs_hs) > 0:
+                cur_vllm_emb = vllm_embedding[i]
+                cur_image_bound = data['image_bound'][i]
+                if len(cur_image_bound) > 0:
+                    image_indices = torch.stack(
+                        [torch.arange(r[0], r[1], dtype=torch.long) for r in cur_image_bound]
+                    ).to(vllm_embedding.device)
+
+                    cur_vllm_emb.scatter_(0, image_indices.view(-1, 1).repeat(1, cur_vllm_emb.shape[-1]),
+                                          cur_vs_hs.view(-1, cur_vs_hs.shape[-1]))
+                elif self.training:
+                    cur_vllm_emb += cur_vs_hs[0].mean() * 0
+
+        return vllm_embedding, vision_hidden_states
+
+    def forward(self, data, **kwargs):
+        vllm_embedding, vision_hidden_states = self.get_vllm_embedding(data)
+        position_ids = data["position_ids"]
+        if position_ids.dtype != torch.int64:
+            position_ids = position_ids.long()
+
+        return self.llm(
+            input_ids=None,
+            position_ids=position_ids,
+            inputs_embeds=vllm_embedding,
+            **kwargs
+        )
+
+    def _decode_text(self, result_ids, tokenizer):
+        result_text = []
+        for result in result_ids:
+            result = result[result != 0]
+            if result[0] == tokenizer.bos_id:
+                result = result[1:]
+            if result[-1] == tokenizer.eos_id or result[-1] == tokenizer.eot_id:
+                result = result[:-1]
+            result_text.append(tokenizer.decode(result).strip())
+        return result_text
+
+    def _decode(self, inputs_embeds, tokenizer, decode_text=False, **kwargs):
+        terminators = [
+            tokenizer.eos_token_id,
+            tokenizer.convert_tokens_to_ids("<|eot_id|>")
+        ]
+        output = self.llm.generate(
+            inputs_embeds=inputs_embeds,
+            pad_token_id=0,
+            eos_token_id=terminators,
+            **kwargs
+        )
+        if decode_text:
+            return self._decode_text(output, tokenizer)
+        return output
+
+    def _decode_stream(self, inputs_embeds, tokenizer, **kwargs):
+        terminators = [
+            tokenizer.eos_token_id,
+            tokenizer.convert_tokens_to_ids("<|eot_id|>")
+        ]
+        streamer = TextIteratorStreamer(tokenizer=tokenizer)
+        generation_kwargs = {
+            'inputs_embeds': inputs_embeds,
+            'pad_token_id': 0,
+            'eos_token_id': terminators,
+            'streamer': streamer
+        }
+        generation_kwargs.update(kwargs)
+
+        thread = Thread(target=self.llm.generate, kwargs=generation_kwargs)
+        thread.start()
+
+        return streamer
+
+    def generate(
+            self,
+            model_inputs,
+            tokenizer=None,
+            vision_hidden_states=None,
+            stream=False,
+            **kwargs
+    ):
+        bs = len(model_inputs["input_ids"])
+        img_list = model_inputs["pixel_values"]
+        tgt_sizes = model_inputs["tgt_sizes"]
+        if img_list is None:
+            img_list = [[] for i in range(bs)]
+        assert bs == len(img_list)
+        if vision_hidden_states is None:
+            pixel_values = []
+            for i in range(bs):
+                img_inps = []
+                for img in img_list[i]:
+                    img_inps.append(img.to(self.device))
+                if img_inps:
+                    pixel_values.append(img_inps)
+                else:
+                    pixel_values.append([])
+            model_inputs["pixel_values"] = pixel_values
+            model_inputs['tgt_sizes'] = tgt_sizes
+        else:
+            model_inputs["vision_hidden_states"] = vision_hidden_states
+
+        (
+            input_embeds,
+            vision_hidden_states,
+        ) = self.get_vllm_embedding(model_inputs)
+
+        # output_ids = self._decode(input_embeds, tokenizer, **kwargs)
+        if stream:
+            kwargs.pop("decode_text")
+            result = self._decode_stream(input_embeds, tokenizer, **kwargs)
+        else:
+            result = self._decode(input_embeds, tokenizer, **kwargs)
+
+        return result
+
+    def chat(
+            self,
+            image,
+            msgs,
+            tokenizer,
+            processor=None,
+            vision_hidden_states=None,
+            max_new_tokens=1024,
+            sampling=True,
+            max_inp_length=2048,
+            system_prompt='',
+            stream=False,
+            **kwargs
+    ):
+        if processor is None:
+            processor = AutoProcessor.from_pretrained(self.config._name_or_path, trust_remote_code=True)
+        if isinstance(msgs, str):
+            msgs = json.loads(msgs)
+        copy_msgs = deepcopy(msgs)
+
+        assert len(msgs) > 0, "msgs is empty"
+        assert sampling or not stream, "if use stream mode, make sure sampling=True"
+
+        if image is not None and isinstance(copy_msgs[0]["content"], str):
+            # copy_msgs[0]['content'] = '(<image>./</image>)\n' + copy_msgs[0]['content']
+            copy_msgs[0]["content"] = [image, copy_msgs[0]["content"]]
+
+        images = []
+        for i, msg in enumerate(copy_msgs):
+            role = msg["role"]
+            content = msg["content"]
+            assert role in ["user", "assistant"]
+            if i == 0:
+                assert role == "user", "The role of first msg should be user"
+            if isinstance(content, str):
+                content = [content]
+            cur_msgs = []
+            for c in content:
+                if isinstance(c, Image.Image):
+                    images.append(c)
+                    cur_msgs.append("(<image>./</image>)")
+                elif isinstance(c, str):
+                    cur_msgs.append(c)
+            msg["content"] = "\n".join(cur_msgs)
+
+        if system_prompt:
+            sys_msg = {'role': 'system', 'content': system_prompt}
+            copy_msgs = [sys_msg] + copy_msgs
+
+        prompt = processor.tokenizer.apply_chat_template(copy_msgs, tokenize=False, add_generation_prompt=True)
+        inputs = processor(prompt, images, return_tensors="pt", max_length=max_inp_length).to(self.device)
+
+        if sampling:
+            generation_config = {
+                "top_p": 0.8,
+                "top_k": 100,
+                "temperature": 0.7,
+                "do_sample": True,
+                "repetition_penalty": 1.05
+            }
+        else:
+            generation_config = {
+                "num_beams": 3,
+                "repetition_penalty": 1.2,
+            }
+
+        generation_config.update(
+            (k, kwargs[k]) for k in generation_config.keys() & kwargs.keys()
+        )
+        with torch.inference_mode():
+            res = self.generate(
+                inputs,
+                tokenizer=tokenizer,
+                max_new_tokens=max_new_tokens,
+                vision_hidden_states=vision_hidden_states,
+                stream=stream,
+                decode_text=True,
+                **generation_config
+            )
+
+        if stream:
+            def stream_gen():
+                for text in res:
+                    text = text.replace(tokenizer.eot_token, '').replace(tokenizer.eos_token, '')
+                    yield text
+
+            return stream_gen()
+
+        else:
+            answer = res[0]
+            return answer
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/huggingface_modify/resampler.py b/PyTorch/built-in/mlm/MiniCPM-V/huggingface_modify/resampler.py
new file mode 100644
index 0000000000000000000000000000000000000000..c0d13540ae44d92aa9c5672e5b054d6289b986e0
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/huggingface_modify/resampler.py
@@ -0,0 +1,815 @@
+from functools import partial
+import numpy as np
+import warnings
+from typing import Optional, Tuple
+import torch
+from torch import nn
+from torch import Tensor
+import torch.nn.functional as F
+from torch.nn.functional import *
+from torch.nn.modules.activation import *
+from torch.nn.init import trunc_normal_
+from torch.nn.init import constant_, xavier_normal_, xavier_uniform_
+from transformers import PreTrainedModel
+from transformers.integrations import is_deepspeed_zero3_enabled
+
+
+def get_2d_sincos_pos_embed(embed_dim, image_size):
+    """
+    image_size: image_size or (image_height, image_width)
+    return:
+    pos_embed: [image_height, image_width, embed_dim]
+    """
+    if isinstance(image_size, int):
+        grid_h_size, grid_w_size = image_size, image_size
+    else:
+        grid_h_size, grid_w_size = image_size[0], image_size[1]
+
+    grid_h = np.arange(grid_h_size, dtype=np.float32)
+    grid_w = np.arange(grid_w_size, dtype=np.float32)
+    grid = np.meshgrid(grid_w, grid_h)  # here w goes first
+    grid = np.stack(grid, axis=0)
+
+    pos_embed = get_2d_sincos_pos_embed_from_grid(embed_dim, grid)
+    return pos_embed
+
+
+def get_2d_sincos_pos_embed_from_grid(embed_dim, grid):
+    assert embed_dim % 2 == 0
+
+    # use half of dimensions to encode grid_h
+    emb_h = get_1d_sincos_pos_embed_from_grid_new(embed_dim // 2, grid[0])  # (H, W, D/2)
+    emb_w = get_1d_sincos_pos_embed_from_grid_new(embed_dim // 2, grid[1])  # (H, W, D/2)
+
+    emb = np.concatenate([emb_h, emb_w], axis=-1)  # (H, W, D)
+    return emb
+
+
+def get_1d_sincos_pos_embed_from_grid_new(embed_dim, pos):
+    """
+    embed_dim: output dimension for each position
+    pos: a list of positions to be encoded: size (H, W)
+    out: (H, W, D)
+    """
+    assert embed_dim % 2 == 0
+    omega = np.arange(embed_dim // 2, dtype=np.float32)
+    omega /= embed_dim / 2.
+    omega = 1. / 10000 ** omega  # (D/2,)
+
+    out = np.einsum('hw,d->hwd', pos, omega)  # (H, W, D/2), outer product
+
+    emb_sin = np.sin(out)  # (H, W, D/2)
+    emb_cos = np.cos(out)  # (H, W, D/2)
+
+    emb = np.concatenate([emb_sin, emb_cos], axis=-1)  # (H, W, D)
+    return emb
+
+
+class Resampler(nn.Module):
+    """
+    A 2D perceiver-resampler network with one cross attention layers by
+       given learnable queries and 2d sincos pos_emb
+    Outputs:
+        A tensor with the shape of (batch_size, num_queries, embed_dim)
+    """
+
+    def __init__(
+            self,
+            num_queries,
+            embed_dim,
+            num_heads,
+            kv_dim=None,
+            norm_layer=partial(nn.LayerNorm, eps=1e-6),
+            adaptive=False,
+            max_size=(70, 70),
+    ):
+        super().__init__()
+        self.num_queries = num_queries
+        self.embed_dim = embed_dim
+        self.num_heads = num_heads
+        self.adaptive = adaptive
+        self.max_size = max_size
+
+        self.query = nn.Parameter(torch.zeros(self.num_queries, embed_dim))
+
+        if kv_dim is not None and kv_dim != embed_dim:
+            self.kv_proj = nn.Linear(kv_dim, embed_dim, bias=False)
+        else:
+            self.kv_proj = nn.Identity()
+
+        self.attn = MultiheadAttention(embed_dim, num_heads)
+        self.ln_q = norm_layer(embed_dim)
+        self.ln_kv = norm_layer(embed_dim)
+
+        self.ln_post = norm_layer(embed_dim)
+        self.proj = nn.Parameter((embed_dim ** -0.5) * torch.randn(embed_dim, embed_dim))
+
+        self._set_2d_pos_cache(self.max_size)
+
+    def _set_2d_pos_cache(self, max_size, device='cpu'):
+        if is_deepspeed_zero3_enabled():
+            device = 'cuda'
+        pos_embed = torch.from_numpy(get_2d_sincos_pos_embed(self.embed_dim, max_size)).float().to(device)
+        self.register_buffer("pos_embed", pos_embed, persistent=False)
+
+    def _adjust_pos_cache(self, tgt_sizes, device):
+        max_h = torch.max(tgt_sizes[:, 0])
+        max_w = torch.max(tgt_sizes[:, 1])
+        if max_h > self.max_size[0] or max_w > self.max_size[1]:
+            self.max_size = [max(max_h, self.max_size[0]), max(max_w, self.max_size[1])]
+            self._set_2d_pos_cache(self.max_size, device)
+
+    def _init_weights(self, m):
+        if isinstance(m, nn.Linear):
+            trunc_normal_(m.weight, std=.02)
+            if isinstance(m, nn.Linear) and m.bias is not None:
+                nn.init.constant_(m.bias, 0)
+        elif isinstance(m, nn.LayerNorm):
+            nn.init.constant_(m.bias, 0)
+            nn.init.constant_(m.weight, 1.0)
+
+    def forward(self, x, tgt_sizes=None):
+        assert x.shape[0] == tgt_sizes.shape[0]
+        bs = x.shape[0]
+
+        device = x.device
+        dtype = x.dtype
+
+        patch_len = tgt_sizes[:, 0] * tgt_sizes[:, 1]
+
+        self._adjust_pos_cache(tgt_sizes, device=device)
+
+        max_patch_len = torch.max(patch_len)
+        key_padding_mask = torch.zeros((bs, max_patch_len), dtype=torch.bool, device=device)
+
+        pos_embed = []
+        for i in range(bs):
+            tgt_h, tgt_w = tgt_sizes[i]
+            pos_embed.append(self.pos_embed[:tgt_h, :tgt_w, :].reshape((tgt_h * tgt_w, -1)).to(dtype))  # patches * D
+            key_padding_mask[i, patch_len[i]:] = True
+
+        pos_embed = torch.nn.utils.rnn.pad_sequence(
+            pos_embed, batch_first=True, padding_value=0.0).permute(1, 0, 2)  # BLD => L * B * D
+
+        x = self.kv_proj(x)  # B * L * D
+        x = self.ln_kv(x).permute(1, 0, 2)  # L * B * D
+
+        q = self.ln_q(self.query)  # Q * D
+
+        out = self.attn(
+            self._repeat(q, bs),  # Q * B * D
+            x + pos_embed,  # L * B * D +  L * B * D
+            x,
+            key_padding_mask=key_padding_mask)[0]
+        #  out: Q * B * D
+        x = out.permute(1, 0, 2)  # B * Q * D
+
+        x = self.ln_post(x)
+        x = x @ self.proj
+        return x
+
+    def _repeat(self, query, N: int):
+        return query.unsqueeze(1).repeat(1, N, 1)
+
+
+class MultiheadAttention(nn.MultiheadAttention):
+    def __init__(self, embed_dim, num_heads, dropout=0., bias=True, add_bias_kv=False,
+                 add_zero_attn=False, kdim=None, vdim=None, batch_first=False, device=None, dtype=None):
+        super().__init__(embed_dim, num_heads, dropout, bias, add_bias_kv, add_zero_attn, kdim, vdim, batch_first,
+                         device, dtype)
+
+        # rewrite out_proj layer，with nn.Linear
+        self.out_proj = nn.Linear(embed_dim, embed_dim, bias=bias, device=device, dtype=dtype)
+
+    def forward(
+            self,
+            query: Tensor,
+            key: Tensor,
+            value: Tensor,
+            key_padding_mask: Optional[Tensor] = None,
+            need_weights: bool = True,
+            attn_mask: Optional[Tensor] = None,
+            average_attn_weights: bool = True,
+            is_causal: bool = False) -> Tuple[Tensor, Optional[Tensor]]:
+        why_not_fast_path = ''
+        if ((attn_mask is not None and torch.is_floating_point(attn_mask))
+                or (key_padding_mask is not None) and torch.is_floating_point(key_padding_mask)):
+            why_not_fast_path = "floating-point masks are not supported for fast path."
+
+        is_batched = query.dim() == 3
+
+        key_padding_mask = F._canonical_mask(
+            mask=key_padding_mask,
+            mask_name="key_padding_mask",
+            other_type=F._none_or_dtype(attn_mask),
+            other_name="attn_mask",
+            target_type=query.dtype
+        )
+
+        attn_mask = F._canonical_mask(
+            mask=attn_mask,
+            mask_name="attn_mask",
+            other_type=None,
+            other_name="",
+            target_type=query.dtype,
+            check_other=False,
+        )
+
+        if not is_batched:
+            why_not_fast_path = f"input not batched; expected query.dim() of 3 but got {query.dim()}"
+        elif query is not key or key is not value:
+            # When lifting this restriction, don't forget to either
+            # enforce that the dtypes all match or test cases where
+            # they don't!
+            why_not_fast_path = "non-self attention was used (query, key, and value are not the same Tensor)"
+        elif self.in_proj_bias is not None and query.dtype != self.in_proj_bias.dtype:
+            why_not_fast_path = f"dtypes of query ({query.dtype}) and self.in_proj_bias ({self.in_proj_bias.dtype}) don't match"
+        elif self.in_proj_weight is None:
+            why_not_fast_path = "in_proj_weight was None"
+        elif query.dtype != self.in_proj_weight.dtype:
+            # this case will fail anyway, but at least they'll get a useful error message.
+            why_not_fast_path = f"dtypes of query ({query.dtype}) and self.in_proj_weight ({self.in_proj_weight.dtype}) don't match"
+        elif self.training:
+            why_not_fast_path = "training is enabled"
+        elif (self.num_heads % 2) != 0:
+            why_not_fast_path = "self.num_heads is not even"
+        elif not self.batch_first:
+            why_not_fast_path = "batch_first was not True"
+        elif self.bias_k is not None:
+            why_not_fast_path = "self.bias_k was not None"
+        elif self.bias_v is not None:
+            why_not_fast_path = "self.bias_v was not None"
+        elif self.add_zero_attn:
+            why_not_fast_path = "add_zero_attn was enabled"
+        elif not self._qkv_same_embed_dim:
+            why_not_fast_path = "_qkv_same_embed_dim was not True"
+        elif query.is_nested and (key_padding_mask is not None or attn_mask is not None):
+            why_not_fast_path = "supplying both src_key_padding_mask and src_mask at the same time \
+                                 is not supported with NestedTensor input"
+        elif torch.is_autocast_enabled():
+            why_not_fast_path = "autocast is enabled"
+
+        if not why_not_fast_path:
+            tensor_args = (
+                query,
+                key,
+                value,
+                self.in_proj_weight,
+                self.in_proj_bias,
+                self.out_proj.weight,
+                self.out_proj.bias,
+            )
+            # We have to use list comprehensions below because TorchScript does not support
+            # generator expressions.
+            if torch.overrides.has_torch_function(tensor_args):
+                why_not_fast_path = "some Tensor argument has_torch_function"
+            elif _is_make_fx_tracing():
+                why_not_fast_path = "we are running make_fx tracing"
+            elif not all(_check_arg_device(x) for x in tensor_args):
+                why_not_fast_path = ("some Tensor argument's device is neither one of "
+                                     f"cpu, cuda or {torch.utils.backend_registration._privateuse1_backend_name}")
+            elif torch.is_grad_enabled() and any(_arg_requires_grad(x) for x in tensor_args):
+                why_not_fast_path = ("grad is enabled and at least one of query or the "
+                                     "input/output projection weights or biases requires_grad")
+            if not why_not_fast_path:
+                merged_mask, mask_type = self.merge_masks(attn_mask, key_padding_mask, query)
+
+                if self.in_proj_bias is not None and self.in_proj_weight is not None:
+                    return torch._native_multi_head_attention(
+                        query,
+                        key,
+                        value,
+                        self.embed_dim,
+                        self.num_heads,
+                        self.in_proj_weight,
+                        self.in_proj_bias,
+                        self.out_proj.weight,
+                        self.out_proj.bias,
+                        merged_mask,
+                        need_weights,
+                        average_attn_weights,
+                        mask_type)
+
+        any_nested = query.is_nested or key.is_nested or value.is_nested
+        assert not any_nested, ("MultiheadAttention does not support NestedTensor outside of its fast path. " +
+                                f"The fast path was not hit because {why_not_fast_path}")
+
+        if self.batch_first and is_batched:
+            # make sure that the transpose op does not affect the "is" property
+            if key is value:
+                if query is key:
+                    query = key = value = query.transpose(1, 0)
+                else:
+                    query, key = (x.transpose(1, 0) for x in (query, key))
+                    value = key
+            else:
+                query, key, value = (x.transpose(1, 0) for x in (query, key, value))
+
+        if not self._qkv_same_embed_dim:
+            attn_output, attn_output_weights = self.multi_head_attention_forward(
+                query, key, value, self.embed_dim, self.num_heads,
+                self.in_proj_weight, self.in_proj_bias,
+                self.bias_k, self.bias_v, self.add_zero_attn,
+                self.dropout, self.out_proj.weight, self.out_proj.bias,
+                training=self.training,
+                key_padding_mask=key_padding_mask, need_weights=need_weights,
+                attn_mask=attn_mask,
+                use_separate_proj_weight=True,
+                q_proj_weight=self.q_proj_weight, k_proj_weight=self.k_proj_weight,
+                v_proj_weight=self.v_proj_weight,
+                average_attn_weights=average_attn_weights,
+                is_causal=is_causal)
+        else:
+            attn_output, attn_output_weights = self.multi_head_attention_forward(
+                query, key, value, self.embed_dim, self.num_heads,
+                self.in_proj_weight, self.in_proj_bias,
+                self.bias_k, self.bias_v, self.add_zero_attn,
+                self.dropout, self.out_proj.weight, self.out_proj.bias,
+                training=self.training,
+                key_padding_mask=key_padding_mask,
+                need_weights=need_weights,
+                attn_mask=attn_mask,
+                average_attn_weights=average_attn_weights,
+                is_causal=is_causal)
+        if self.batch_first and is_batched:
+            return attn_output.transpose(1, 0), attn_output_weights
+        else:
+            return attn_output, attn_output_weights
+
+    def multi_head_attention_forward(
+            self,
+            query: Tensor,
+            key: Tensor,
+            value: Tensor,
+            embed_dim_to_check: int,
+            num_heads: int,
+            in_proj_weight: Optional[Tensor],
+            in_proj_bias: Optional[Tensor],
+            bias_k: Optional[Tensor],
+            bias_v: Optional[Tensor],
+            add_zero_attn: bool,
+            dropout_p: float,
+            out_proj_weight: Tensor,
+            out_proj_bias: Optional[Tensor],
+            training: bool = True,
+            key_padding_mask: Optional[Tensor] = None,
+            need_weights: bool = True,
+            attn_mask: Optional[Tensor] = None,
+            use_separate_proj_weight: bool = False,
+            q_proj_weight: Optional[Tensor] = None,
+            k_proj_weight: Optional[Tensor] = None,
+            v_proj_weight: Optional[Tensor] = None,
+            static_k: Optional[Tensor] = None,
+            static_v: Optional[Tensor] = None,
+            average_attn_weights: bool = True,
+            is_causal: bool = False,
+    ) -> Tuple[Tensor, Optional[Tensor]]:
+        tens_ops = (query, key, value, in_proj_weight, in_proj_bias, bias_k, bias_v, out_proj_weight, out_proj_bias)
+        if has_torch_function(tens_ops):
+            return handle_torch_function(
+                multi_head_attention_forward,
+                tens_ops,
+                query,
+                key,
+                value,
+                embed_dim_to_check,
+                num_heads,
+                in_proj_weight,
+                in_proj_bias,
+                bias_k,
+                bias_v,
+                add_zero_attn,
+                dropout_p,
+                out_proj_weight,
+                out_proj_bias,
+                training=training,
+                key_padding_mask=key_padding_mask,
+                need_weights=need_weights,
+                attn_mask=attn_mask,
+                is_causal=is_causal,
+                use_separate_proj_weight=use_separate_proj_weight,
+                q_proj_weight=q_proj_weight,
+                k_proj_weight=k_proj_weight,
+                v_proj_weight=v_proj_weight,
+                static_k=static_k,
+                static_v=static_v,
+                average_attn_weights=average_attn_weights,
+            )
+
+        is_batched = _mha_shape_check(query, key, value, key_padding_mask, attn_mask, num_heads)
+
+        # For unbatched input, we unsqueeze at the expected batch-dim to pretend that the input
+        # is batched, run the computation and before returning squeeze the
+        # batch dimension so that the output doesn't carry this temporary batch dimension.
+        if not is_batched:
+            # unsqueeze if the input is unbatched
+            query = query.unsqueeze(1)
+            key = key.unsqueeze(1)
+            value = value.unsqueeze(1)
+            if key_padding_mask is not None:
+                key_padding_mask = key_padding_mask.unsqueeze(0)
+
+        # set up shape vars
+        tgt_len, bsz, embed_dim = query.shape
+        src_len, _, _ = key.shape
+
+        key_padding_mask = _canonical_mask(
+            mask=key_padding_mask,
+            mask_name="key_padding_mask",
+            other_type=_none_or_dtype(attn_mask),
+            other_name="attn_mask",
+            target_type=query.dtype
+        )
+
+        if is_causal and attn_mask is None:
+            raise RuntimeError(
+                "Need attn_mask if specifying the is_causal hint. "
+                "You may use the Transformer module method "
+                "`generate_square_subsequent_mask` to create this mask."
+            )
+
+        if is_causal and key_padding_mask is None and not need_weights:
+            # when we have a kpm or need weights, we need attn_mask
+            # Otherwise, we use the is_causal hint go as is_causal
+            # indicator to SDPA.
+            attn_mask = None
+        else:
+            attn_mask = _canonical_mask(
+                mask=attn_mask,
+                mask_name="attn_mask",
+                other_type=None,
+                other_name="",
+                target_type=query.dtype,
+                check_other=False,
+            )
+
+            if key_padding_mask is not None:
+                # We have the attn_mask, and use that to merge kpm into it.
+                # Turn off use of is_causal hint, as the merged mask is no
+                # longer causal.
+                is_causal = False
+
+        assert embed_dim == embed_dim_to_check, \
+            f"was expecting embedding dimension of {embed_dim_to_check}, but got {embed_dim}"
+        if isinstance(embed_dim, torch.Tensor):
+            # embed_dim can be a tensor when JIT tracing
+            head_dim = embed_dim.div(num_heads, rounding_mode='trunc')
+        else:
+            head_dim = embed_dim // num_heads
+        assert head_dim * num_heads == embed_dim, f"embed_dim {embed_dim} not divisible by num_heads {num_heads}"
+        if use_separate_proj_weight:
+            # allow MHA to have different embedding dimensions when separate projection weights are used
+            assert key.shape[:2] == value.shape[:2], \
+                f"key's sequence and batch dims {key.shape[:2]} do not match value's {value.shape[:2]}"
+        else:
+            assert key.shape == value.shape, f"key shape {key.shape} does not match value shape {value.shape}"
+
+        #
+        # compute in-projection
+        #
+        if not use_separate_proj_weight:
+            assert in_proj_weight is not None, "use_separate_proj_weight is False but in_proj_weight is None"
+            q, k, v = _in_projection_packed(query, key, value, in_proj_weight, in_proj_bias)
+        else:
+            assert q_proj_weight is not None, "use_separate_proj_weight is True but q_proj_weight is None"
+            assert k_proj_weight is not None, "use_separate_proj_weight is True but k_proj_weight is None"
+            assert v_proj_weight is not None, "use_separate_proj_weight is True but v_proj_weight is None"
+            if in_proj_bias is None:
+                b_q = b_k = b_v = None
+            else:
+                b_q, b_k, b_v = in_proj_bias.chunk(3)
+            q, k, v = _in_projection(query, key, value, q_proj_weight, k_proj_weight, v_proj_weight, b_q, b_k, b_v)
+
+        # prep attention mask
+
+        if attn_mask is not None:
+            # ensure attn_mask's dim is 3
+            if attn_mask.dim() == 2:
+                correct_2d_size = (tgt_len, src_len)
+                if attn_mask.shape != correct_2d_size:
+                    raise RuntimeError(
+                        f"The shape of the 2D attn_mask is {attn_mask.shape}, but should be {correct_2d_size}.")
+                attn_mask = attn_mask.unsqueeze(0)
+            elif attn_mask.dim() == 3:
+                correct_3d_size = (bsz * num_heads, tgt_len, src_len)
+                if attn_mask.shape != correct_3d_size:
+                    raise RuntimeError(
+                        f"The shape of the 3D attn_mask is {attn_mask.shape}, but should be {correct_3d_size}.")
+            else:
+                raise RuntimeError(f"attn_mask's dimension {attn_mask.dim()} is not supported")
+
+        # add bias along batch dimension (currently second)
+        if bias_k is not None and bias_v is not None:
+            assert static_k is None, "bias cannot be added to static key."
+            assert static_v is None, "bias cannot be added to static value."
+            k = torch.cat([k, bias_k.repeat(1, bsz, 1)])
+            v = torch.cat([v, bias_v.repeat(1, bsz, 1)])
+            if attn_mask is not None:
+                attn_mask = pad(attn_mask, (0, 1))
+            if key_padding_mask is not None:
+                key_padding_mask = pad(key_padding_mask, (0, 1))
+        else:
+            assert bias_k is None
+            assert bias_v is None
+
+        #
+        # reshape q, k, v for multihead attention and make em batch first
+        #
+        q = q.view(tgt_len, bsz * num_heads, head_dim).transpose(0, 1)
+        if static_k is None:
+            k = k.view(k.shape[0], bsz * num_heads, head_dim).transpose(0, 1)
+        else:
+            # TODO finish disentangling control flow so we don't do in-projections when statics are passed
+            assert static_k.size(0) == bsz * num_heads, \
+                f"expecting static_k.size(0) of {bsz * num_heads}, but got {static_k.size(0)}"
+            assert static_k.size(2) == head_dim, \
+                f"expecting static_k.size(2) of {head_dim}, but got {static_k.size(2)}"
+            k = static_k
+        if static_v is None:
+            v = v.view(v.shape[0], bsz * num_heads, head_dim).transpose(0, 1)
+        else:
+            # TODO finish disentangling control flow so we don't do in-projections when statics are passed
+            assert static_v.size(0) == bsz * num_heads, \
+                f"expecting static_v.size(0) of {bsz * num_heads}, but got {static_v.size(0)}"
+            assert static_v.size(2) == head_dim, \
+                f"expecting static_v.size(2) of {head_dim}, but got {static_v.size(2)}"
+            v = static_v
+
+        # add zero attention along batch dimension (now first)
+        if add_zero_attn:
+            zero_attn_shape = (bsz * num_heads, 1, head_dim)
+            k = torch.cat([k, torch.zeros(zero_attn_shape, dtype=k.dtype, device=k.device)], dim=1)
+            v = torch.cat([v, torch.zeros(zero_attn_shape, dtype=v.dtype, device=v.device)], dim=1)
+            if attn_mask is not None:
+                attn_mask = pad(attn_mask, (0, 1))
+            if key_padding_mask is not None:
+                key_padding_mask = pad(key_padding_mask, (0, 1))
+
+        # update source sequence length after adjustments
+        src_len = k.size(1)
+
+        # merge key padding and attention masks
+        if key_padding_mask is not None:
+            assert key_padding_mask.shape == (bsz, src_len), \
+                f"expecting key_padding_mask shape of {(bsz, src_len)}, but got {key_padding_mask.shape}"
+            key_padding_mask = key_padding_mask.view(bsz, 1, 1, src_len). \
+                expand(-1, num_heads, -1, -1).reshape(bsz * num_heads, 1, src_len)
+            if attn_mask is None:
+                attn_mask = key_padding_mask
+            else:
+                attn_mask = attn_mask + key_padding_mask
+
+        # adjust dropout probability
+        if not training:
+            dropout_p = 0.0
+
+        #
+        # (deep breath) calculate attention and out projection
+        #
+
+        if need_weights:
+            B, Nt, E = q.shape
+            q_scaled = q / math.sqrt(E)
+
+            assert not (is_causal and attn_mask is None), "FIXME: is_causal not implemented for need_weights"
+
+            if attn_mask is not None:
+                attn_output_weights = torch.baddbmm(attn_mask, q_scaled, k.transpose(-2, -1))
+            else:
+                attn_output_weights = torch.bmm(q_scaled, k.transpose(-2, -1))
+            attn_output_weights = softmax(attn_output_weights, dim=-1)
+            if dropout_p > 0.0:
+                attn_output_weights = dropout(attn_output_weights, p=dropout_p)
+
+            attn_output = torch.bmm(attn_output_weights, v)
+
+            attn_output = attn_output.transpose(0, 1).contiguous().view(tgt_len * bsz, embed_dim)
+            attn_output = self.out_proj(attn_output)
+            attn_output = attn_output.view(tgt_len, bsz, attn_output.size(1))
+
+            # optionally average attention weights over heads
+            attn_output_weights = attn_output_weights.view(bsz, num_heads, tgt_len, src_len)
+            if average_attn_weights:
+                attn_output_weights = attn_output_weights.mean(dim=1)
+
+            if not is_batched:
+                # squeeze the output if input was unbatched
+                attn_output = attn_output.squeeze(1)
+                attn_output_weights = attn_output_weights.squeeze(0)
+            return attn_output, attn_output_weights
+        else:
+            # attn_mask can be either (L,S) or (N*num_heads, L, S)
+            # if attn_mask's shape is (1, L, S) we need to unsqueeze to (1, 1, L, S)
+            # in order to match the input for SDPA of (N, num_heads, L, S)
+            if attn_mask is not None:
+                if attn_mask.size(0) == 1 and attn_mask.dim() == 3:
+                    attn_mask = attn_mask.unsqueeze(0)
+                else:
+                    attn_mask = attn_mask.view(bsz, num_heads, -1, src_len)
+
+            q = q.view(bsz, num_heads, tgt_len, head_dim)
+            k = k.view(bsz, num_heads, src_len, head_dim)
+            v = v.view(bsz, num_heads, src_len, head_dim)
+
+            attn_output = F.scaled_dot_product_attention(q, k, v, attn_mask, dropout_p, is_causal)
+            attn_output = attn_output.permute(2, 0, 1, 3).contiguous().view(bsz * tgt_len, embed_dim)
+
+            attn_output = self.out_proj(attn_output)
+            attn_output = attn_output.view(tgt_len, bsz, attn_output.size(1))
+            if not is_batched:
+                # squeeze the output if input was unbatched
+                attn_output = attn_output.squeeze(1)
+            return attn_output, None
+
+
+def _mha_shape_check(query: Tensor, key: Tensor, value: Tensor,
+                     key_padding_mask: Optional[Tensor], attn_mask: Optional[Tensor], num_heads: int):
+    # Verifies the expected shape for `query, `key`, `value`, `key_padding_mask` and `attn_mask`
+    # and returns if the input is batched or not.
+    # Raises an error if `query` is not 2-D (unbatched) or 3-D (batched) tensor.
+
+    # Shape check.
+    if query.dim() == 3:
+        # Batched Inputs
+        is_batched = True
+        assert key.dim() == 3 and value.dim() == 3, \
+            ("For batched (3-D) `query`, expected `key` and `value` to be 3-D"
+             f" but found {key.dim()}-D and {value.dim()}-D tensors respectively")
+        if key_padding_mask is not None:
+            assert key_padding_mask.dim() == 2, \
+                ("For batched (3-D) `query`, expected `key_padding_mask` to be `None` or 2-D"
+                 f" but found {key_padding_mask.dim()}-D tensor instead")
+        if attn_mask is not None:
+            assert attn_mask.dim() in (2, 3), \
+                ("For batched (3-D) `query`, expected `attn_mask` to be `None`, 2-D or 3-D"
+                 f" but found {attn_mask.dim()}-D tensor instead")
+    elif query.dim() == 2:
+        # Unbatched Inputs
+        is_batched = False
+        assert key.dim() == 2 and value.dim() == 2, \
+            ("For unbatched (2-D) `query`, expected `key` and `value` to be 2-D"
+             f" but found {key.dim()}-D and {value.dim()}-D tensors respectively")
+
+        if key_padding_mask is not None:
+            assert key_padding_mask.dim() == 1, \
+                ("For unbatched (2-D) `query`, expected `key_padding_mask` to be `None` or 1-D"
+                 f" but found {key_padding_mask.dim()}-D tensor instead")
+
+        if attn_mask is not None:
+            assert attn_mask.dim() in (2, 3), \
+                ("For unbatched (2-D) `query`, expected `attn_mask` to be `None`, 2-D or 3-D"
+                 f" but found {attn_mask.dim()}-D tensor instead")
+            if attn_mask.dim() == 3:
+                expected_shape = (num_heads, query.shape[0], key.shape[0])
+                assert attn_mask.shape == expected_shape, \
+                    (f"Expected `attn_mask` shape to be {expected_shape} but got {attn_mask.shape}")
+    else:
+        raise AssertionError(
+            f"query should be unbatched 2D or batched 3D tensor but received {query.dim()}-D query tensor")
+
+    return is_batched
+
+
+def _canonical_mask(
+        mask: Optional[Tensor],
+        mask_name: str,
+        other_type: Optional[DType],
+        other_name: str,
+        target_type: DType,
+        check_other: bool = True,
+) -> Optional[Tensor]:
+    if mask is not None:
+        _mask_dtype = mask.dtype
+        _mask_is_float = torch.is_floating_point(mask)
+        if _mask_dtype != torch.bool and not _mask_is_float:
+            raise AssertionError(
+                f"only bool and floating types of {mask_name} are supported")
+        if check_other and other_type is not None:
+            if _mask_dtype != other_type:
+                warnings.warn(
+                    f"Support for mismatched {mask_name} and {other_name} "
+                    "is deprecated. Use same type for both instead."
+                )
+        if not _mask_is_float:
+            mask = (
+                torch.zeros_like(mask, dtype=target_type)
+                .masked_fill_(mask, float("-inf"))
+            )
+    return mask
+
+
+def _none_or_dtype(input: Optional[Tensor]) -> Optional[DType]:
+    if input is None:
+        return None
+    elif isinstance(input, torch.Tensor):
+        return input.dtype
+    raise RuntimeError("input to _none_or_dtype() must be None or torch.Tensor")
+
+
+def _in_projection_packed(
+        q: Tensor,
+        k: Tensor,
+        v: Tensor,
+        w: Tensor,
+        b: Optional[Tensor] = None,
+) -> List[Tensor]:
+    r"""
+    Performs the in-projection step of the attention operation, using packed weights.
+    Output is a triple containing projection tensors for query, key and value.
+    Args:
+        q, k, v: query, key and value tensors to be projected. For self-attention,
+            these are typically the same tensor; for encoder-decoder attention,
+            k and v are typically the same tensor. (We take advantage of these
+            identities for performance if they are present.) Regardless, q, k and v
+            must share a common embedding dimension; otherwise their shapes may vary.
+        w: projection weights for q, k and v, packed into a single tensor. Weights
+            are packed along dimension 0, in q, k, v order.
+        b: optional projection biases for q, k and v, packed into a single tensor
+            in q, k, v order.
+    Shape:
+        Inputs:
+        - q: :math:`(..., E)` where E is the embedding dimension
+        - k: :math:`(..., E)` where E is the embedding dimension
+        - v: :math:`(..., E)` where E is the embedding dimension
+        - w: :math:`(E * 3, E)` where E is the embedding dimension
+        - b: :math:`E * 3` where E is the embedding dimension
+        Output:
+        - in output list :math:`[q', k', v']`, each output tensor will have the
+            same shape as the corresponding input tensor.
+    """
+    E = q.size(-1)
+    if k is v:
+        if q is k:
+            # self-attention
+            proj = linear(q, w, b)
+            # reshape to 3, E and not E, 3 is deliberate for better memory coalescing and keeping same order as chunk()
+            proj = proj.unflatten(-1, (3, E)).unsqueeze(0).transpose(0, -2).squeeze(-2).contiguous()
+            return proj[0], proj[1], proj[2]
+        else:
+            # encoder-decoder attention
+            w_q, w_kv = w.split([E, E * 2])
+            if b is None:
+                b_q = b_kv = None
+            else:
+                b_q, b_kv = b.split([E, E * 2])
+            q_proj = linear(q, w_q, b_q)
+            kv_proj = linear(k, w_kv, b_kv)
+            # reshape to 2, E and not E, 2 is deliberate for better memory coalescing and keeping same order as chunk()
+            kv_proj = kv_proj.unflatten(-1, (2, E)).unsqueeze(0).transpose(0, -2).squeeze(-2).contiguous()
+            return (q_proj, kv_proj[0], kv_proj[1])
+    else:
+        w_q, w_k, w_v = w.chunk(3)
+        if b is None:
+            b_q = b_k = b_v = None
+        else:
+            b_q, b_k, b_v = b.chunk(3)
+        return linear(q, w_q, b_q), linear(k, w_k, b_k), linear(v, w_v, b_v)
+
+
+def _in_projection(
+        q: Tensor,
+        k: Tensor,
+        v: Tensor,
+        w_q: Tensor,
+        w_k: Tensor,
+        w_v: Tensor,
+        b_q: Optional[Tensor] = None,
+        b_k: Optional[Tensor] = None,
+        b_v: Optional[Tensor] = None,
+) -> Tuple[Tensor, Tensor, Tensor]:
+    r"""
+    Performs the in-projection step of the attention operation. This is simply
+    a triple of linear projections, with shape constraints on the weights which
+    ensure embedding dimension uniformity in the projected outputs.
+    Output is a triple containing projection tensors for query, key and value.
+    Args:
+        q, k, v: query, key and value tensors to be projected.
+        w_q, w_k, w_v: weights for q, k and v, respectively.
+        b_q, b_k, b_v: optional biases for q, k and v, respectively.
+    Shape:
+        Inputs:
+        - q: :math:`(Qdims..., Eq)` where Eq is the query embedding dimension and Qdims are any
+            number of leading dimensions.
+        - k: :math:`(Kdims..., Ek)` where Ek is the key embedding dimension and Kdims are any
+            number of leading dimensions.
+        - v: :math:`(Vdims..., Ev)` where Ev is the value embedding dimension and Vdims are any
+            number of leading dimensions.
+        - w_q: :math:`(Eq, Eq)`
+        - w_k: :math:`(Eq, Ek)`
+        - w_v: :math:`(Eq, Ev)`
+        - b_q: :math:`(Eq)`
+        - b_k: :math:`(Eq)`
+        - b_v: :math:`(Eq)`
+        Output: in output triple :math:`(q', k', v')`,
+         - q': :math:`[Qdims..., Eq]`
+         - k': :math:`[Kdims..., Eq]`
+         - v': :math:`[Vdims..., Eq]`
+    """
+    Eq, Ek, Ev = q.size(-1), k.size(-1), v.size(-1)
+    assert w_q.shape == (Eq, Eq), f"expecting query weights shape of {(Eq, Eq)}, but got {w_q.shape}"
+    assert w_k.shape == (Eq, Ek), f"expecting key weights shape of {(Eq, Ek)}, but got {w_k.shape}"
+    assert w_v.shape == (Eq, Ev), f"expecting value weights shape of {(Eq, Ev)}, but got {w_v.shape}"
+    assert b_q is None or b_q.shape == (Eq,), f"expecting query bias shape of {(Eq,)}, but got {b_q.shape}"
+    assert b_k is None or b_k.shape == (Eq,), f"expecting key bias shape of {(Eq,)}, but got {b_k.shape}"
+    assert b_v is None or b_v.shape == (Eq,), f"expecting value bias shape of {(Eq,)}, but got {b_v.shape}"
+    return linear(q, w_q, b_q), linear(k, w_k, b_k), linear(v, w_v, b_v)
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/npu_monkey_patch/idefics2_conv_monkey_patch.py b/PyTorch/built-in/mlm/MiniCPM-V/npu_monkey_patch/idefics2_conv_monkey_patch.py
new file mode 100644
index 0000000000000000000000000000000000000000..a752532d39f8902e6255e986adb13203a4c4f7e4
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/npu_monkey_patch/idefics2_conv_monkey_patch.py
@@ -0,0 +1,61 @@
+import torch
+from torch import nn
+from transformers.models.idefics2.configuration_idefics2 import Idefics2VisionConfig
+
+
+class Idefics2VisionEmbeddings(nn.Module):
+    """
+    This is a modified version of `siglip.modelign_siglip.SiglipVisionEmbeddings` to enable images of variable
+    resolution.
+
+    The modifications are adapted from [Patch n' Pack: NaViT, a Vision Transformer for any Aspect Ratio and Resolution](https://arxiv.org/abs/2307.06304)
+    which allows treating images in their native aspect ratio and without the need to resize them to the same
+    fixed size. In particular, we start from the original pre-trained SigLIP model
+    (which uses images of fixed-size square images) and adapt it by training on images of variable resolutions.
+    """
+
+    def __init__(self, config: Idefics2VisionConfig):
+        super().__init__()
+        self.embed_dim = config.hidden_size
+        self.image_size = config.image_size
+        self.patch_size = config.patch_size
+
+        self.patch_embedding = nn.Conv2d(
+            in_channels=config.num_channels,
+            out_channels=self.embed_dim,
+            kernel_size=self.patch_size,
+            stride=self.patch_size,
+            padding="valid",
+        )
+
+        self.num_patches_per_side = self.image_size // self.patch_size
+        self.num_patches = self.num_patches_per_side**2
+        self.num_positions = self.num_patches
+        self.position_embedding = nn.Embedding(self.num_positions, self.embed_dim)
+
+    def forward(self, pixel_values: torch.FloatTensor, patch_attention_mask: torch.BoolTensor) -> torch.Tensor:
+        batch_size, _, max_im_h, max_im_w = pixel_values.shape
+
+        patch_embeds = self.patch_embedding(pixel_values)
+        embeddings = patch_embeds.flatten(2).transpose(1, 2)
+
+        max_nb_patches_h, max_nb_patches_w = max_im_h // self.patch_size, max_im_w // self.patch_size
+        boundaries = torch.arange(1 / self.num_patches_per_side, 1.0, 1 / self.num_patches_per_side)
+        position_ids = torch.full(size=(batch_size, max_nb_patches_h * max_nb_patches_w), fill_value=0)
+
+        for batch_idx, p_attn_mask in enumerate(patch_attention_mask):
+            nb_patches_h = p_attn_mask[:, 0].sum()
+            nb_patches_w = p_attn_mask[0].sum()
+
+            fractional_coords_h = torch.arange(0, 1 - 1e-6, 1 / nb_patches_h)
+            fractional_coords_w = torch.arange(0, 1 - 1e-6, 1 / nb_patches_w)
+
+            bucket_coords_h = torch.bucketize(fractional_coords_h, boundaries, right=True)
+            bucket_coords_w = torch.bucketize(fractional_coords_w, boundaries, right=True)
+
+            pos_ids = (bucket_coords_h[:, None] * self.num_patches_per_side + bucket_coords_w).flatten()
+            position_ids[batch_idx][p_attn_mask.view(-1).cpu()] = pos_ids
+
+        position_ids = position_ids.to(self.position_embedding.weight.device)
+        embeddings = embeddings + self.position_embedding(position_ids)
+        return embeddings
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/npu_monkey_patch/idefics2_flash_attn_monkey_patch.py b/PyTorch/built-in/mlm/MiniCPM-V/npu_monkey_patch/idefics2_flash_attn_monkey_patch.py
new file mode 100644
index 0000000000000000000000000000000000000000..010aedbca0d8d5283da0781f4caf7933bb46a552
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/npu_monkey_patch/idefics2_flash_attn_monkey_patch.py
@@ -0,0 +1,204 @@
+from typing import Optional, Tuple
+import torch
+import torch.utils.checkpoint
+import torch_npu
+from npu_monkey_patch.utils import index_first_axis, pad_input, unpad_input
+import transformers
+from transformers.cache_utils import Cache
+from transformers.utils import is_flash_attn_greater_or_equal_2_10, logging, is_flash_attn_2_available
+from transformers.models.idefics2.modeling_idefics2 import Idefics2VisionAttention, _get_unpad_data
+if is_flash_attn_2_available():
+    from flash_attn import flash_attn_func, flash_attn_varlen_func
+
+logger = logging.get_logger(__name__)
+
+
+class Idefics2VisionFlashAttention2(Idefics2VisionAttention):
+    """
+    Idefics2Vision flash attention module. This module inherits from `Idefics2VisionAttention` as the weights of the module stays
+    untouched. The only required change would be on the forward pass where it needs to correctly call the public API of
+    flash attention and deal with padding tokens in case the input contains any of them.
+    """
+
+    # Copied from transformers.models.llama.modeling_llama.LlamaFlashAttention2.__init__
+    def __init__(self, *args, **kwargs):
+        super().__init__(*args, **kwargs)
+
+        # TODO: Should be removed once Flash Attention for RoCm is bumped to 2.1.
+        # flash_attn<2.1 generates top-left aligned causal mask, while what is needed here is bottom-right alignement, that was made default for flash_attn>=2.1. This attribute is used to handle this difference. Reference: https://github.com/Dao-AILab/flash-attention/releases/tag/v2.1.0.
+        # Beware that with flash_attn<2.1, using q_seqlen != k_seqlen (except for the case q_seqlen == 1) produces a wrong mask (top-left).
+        self._flash_attn_uses_top_left_mask = not is_flash_attn_greater_or_equal_2_10()
+
+    def forward(
+        self,
+        hidden_states: torch.Tensor,
+        attention_mask: Optional[torch.LongTensor] = None,
+        position_ids: Optional[torch.LongTensor] = None,
+        past_key_value: Optional[Cache] = None,
+        output_attentions: bool = False,
+        use_cache: bool = False,
+        **kwargs,
+    ) -> Tuple[torch.Tensor, Optional[torch.Tensor], Optional[Tuple[torch.Tensor]]]:
+        output_attentions = False
+
+        bsz, q_len, _ = hidden_states.size()
+
+        query_states = self.q_proj(hidden_states)
+        key_states = self.k_proj(hidden_states)
+        value_states = self.v_proj(hidden_states)
+
+        # Flash attention requires the input to have the shape
+        # batch_size x seq_length x head_dim x hidden_dim
+        # therefore we just need to keep the original shape
+        query_states = query_states.view(bsz, q_len, self.num_heads, self.head_dim).transpose(1, 2)
+        key_states = key_states.view(bsz, q_len, self.num_heads, self.head_dim).transpose(1, 2)
+        value_states = value_states.view(bsz, q_len, self.num_heads, self.head_dim).transpose(1, 2)
+
+        kv_seq_len = key_states.shape[-2]
+        if past_key_value is not None:
+            kv_seq_len += past_key_value.get_usable_length(kv_seq_len, self.layer_idx)
+
+        # TODO: These transpose are quite inefficient but Flash Attention requires the layout [batch_size, sequence_length, num_heads, head_dim]. We would need to refactor the KV cache
+        # to be able to avoid many of these transpose/reshape/view.
+        query_states = query_states.transpose(1, 2)
+        key_states = key_states.transpose(1, 2)
+        value_states = value_states.transpose(1, 2)
+
+        dropout_rate = self.dropout if self.training else 0.0
+
+        # In PEFT, usually we cast the layer norms in float32 for training stability reasons
+        # therefore the input hidden states gets silently casted in float32. Hence, we need
+        # cast them back in the correct dtype just to be sure everything works as expected.
+        # This might slowdown training & inference so it is recommended to not cast the LayerNorms
+        # in fp32. (Idefics2VisionRMSNorm handles it correctly)
+
+        input_dtype = query_states.dtype
+        if input_dtype == torch.float32:
+            if torch.is_autocast_enabled():
+                target_dtype = torch.get_autocast_gpu_dtype()
+            # Handle the case where the model is quantized
+            elif hasattr(self.config, "_pre_quantization_dtype"):
+                target_dtype = self.config._pre_quantization_dtype
+            else:
+                target_dtype = self.q_proj.weight.dtype
+
+            logger.warning_once(
+                f"The input hidden states seems to be silently casted in float32, this might be related to"
+                f" the fact you have upcasted embedding or layer norm layers in float32. We will cast back the input in"
+                f" {target_dtype}."
+            )
+
+            query_states = query_states.to(target_dtype)
+            key_states = key_states.to(target_dtype)
+            value_states = value_states.to(target_dtype)
+
+        attn_output = self._flash_attention_forward(
+            query_states, key_states, value_states, attention_mask, q_len, dropout=dropout_rate
+        )
+
+        attn_output = attn_output.reshape(bsz, q_len, self.embed_dim).contiguous()
+        attn_output = self.out_proj(attn_output)
+
+        if not output_attentions:
+            attn_weights = None
+
+        return attn_output, attn_weights
+
+    # Copied from transformers.models.llama.modeling_llama.LlamaFlashAttention2._flash_attention_forward
+    def _flash_attention_forward(
+        self, query_states, key_states, value_states, attention_mask, query_length, dropout=0.0, softmax_scale=None
+    ):
+        """
+        Calls the forward method of Flash Attention - if the input hidden states contain at least one padding token
+        first unpad the input, then computes the attention scores and pad the final attention scores.
+
+        Args:
+            query_states (`torch.Tensor`):
+                Input query states to be passed to Flash Attention API
+            key_states (`torch.Tensor`):
+                Input key states to be passed to Flash Attention API
+            value_states (`torch.Tensor`):
+                Input value states to be passed to Flash Attention API
+            attention_mask (`torch.Tensor`):
+                The padding mask - corresponds to a tensor of size `(batch_size, seq_len)` where 0 stands for the
+                position of padding tokens and 1 for the position of non-padding tokens.
+            dropout (`float`):
+                Attention dropout
+            softmax_scale (`float`, *optional*):
+                The scaling of QK^T before applying softmax. Default to 1 / sqrt(head_dim)
+        """
+        if not self._flash_attn_uses_top_left_mask:
+            causal = self.is_causal
+        else:
+            # TODO: Remove the `query_length != 1` check once Flash Attention for RoCm is bumped to 2.1. For details, please see the comment in LlamaFlashAttention2 __init__.
+            causal = self.is_causal and query_length != 1
+
+        # Contains at least one padding token in the sequence
+        if attention_mask is not None:
+            batch_size = query_states.shape[0]
+            query_states, key_states, value_states, indices_q, cu_seq_lens, max_seq_lens = self._upad_input(
+                query_states, key_states, value_states, attention_mask, query_length
+            )
+
+            cu_seqlens_q, cu_seqlens_k = cu_seq_lens
+            max_seqlen_in_batch_q, max_seqlen_in_batch_k = max_seq_lens
+
+            attn_output_unpad = flash_attn_varlen_func(
+                query_states,
+                key_states,
+                value_states,
+                cu_seqlens_q=cu_seqlens_q,
+                cu_seqlens_k=cu_seqlens_k,
+                max_seqlen_q=max_seqlen_in_batch_q,
+                max_seqlen_k=max_seqlen_in_batch_k,
+                dropout_p=dropout,
+                softmax_scale=softmax_scale,
+                causal=causal,
+            )
+
+            attn_output = pad_input(attn_output_unpad, indices_q, batch_size, query_length)
+        else:
+            attn_output = flash_attn_func(
+                query_states, key_states, value_states, dropout, softmax_scale=softmax_scale, causal=causal
+            )
+
+        return attn_output
+
+    # Copied from transformers.models.llama.modeling_llama.LlamaFlashAttention2._upad_input
+    def _upad_input(self, query_layer, key_layer, value_layer, attention_mask, query_length):
+        indices_k, cu_seqlens_k, max_seqlen_in_batch_k = _get_unpad_data(attention_mask)
+        batch_size, kv_seq_len, num_key_value_heads, head_dim = key_layer.shape
+
+        key_layer = index_first_axis(
+            key_layer.reshape(batch_size * kv_seq_len, num_key_value_heads, head_dim), indices_k
+        )
+        value_layer = index_first_axis(
+            value_layer.reshape(batch_size * kv_seq_len, num_key_value_heads, head_dim), indices_k
+        )
+        if query_length == kv_seq_len:
+            query_layer = index_first_axis(
+                query_layer.reshape(batch_size * kv_seq_len, self.num_heads, head_dim), indices_k
+            )
+            cu_seqlens_q = cu_seqlens_k
+            max_seqlen_in_batch_q = max_seqlen_in_batch_k
+            indices_q = indices_k
+        elif query_length == 1:
+            max_seqlen_in_batch_q = 1
+            cu_seqlens_q = torch.arange(
+                batch_size + 1, dtype=torch.int32, device=query_layer.device
+            )  # There is a memcpy here, that is very bad.
+            indices_q = cu_seqlens_q[:-1]
+            query_layer = query_layer.squeeze(1)
+        else:
+            # The -q_len: slice assumes left padding.
+            attention_mask = attention_mask[:, -query_length:]
+            query_layer, indices_q, cu_seqlens_q, max_seqlen_in_batch_q = unpad_input(query_layer, attention_mask)
+
+        return (
+            query_layer,
+            key_layer,
+            value_layer,
+            indices_q,
+            (cu_seqlens_q, cu_seqlens_k),
+            (max_seqlen_in_batch_q, max_seqlen_in_batch_k),
+        )
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/npu_monkey_patch/llama_flash_attn_monkey_patch.py b/PyTorch/built-in/mlm/MiniCPM-V/npu_monkey_patch/llama_flash_attn_monkey_patch.py
new file mode 100644
index 0000000000000000000000000000000000000000..10e2a643fd26327038dd2bea8a598af79cfd97a4
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/npu_monkey_patch/llama_flash_attn_monkey_patch.py
@@ -0,0 +1,210 @@
+from typing import Optional, Tuple
+import warnings
+import torch
+import torch.nn as nn
+import torch.utils.checkpoint
+import transformers
+from transformers.cache_utils import Cache
+from .utils import index_first_axis, pad_input, unpad_input
+from transformers.models.llama.modeling_llama import LlamaAttention, apply_rotary_pos_emb, _get_unpad_data
+from transformers.utils import is_flash_attn_greater_or_equal_2_10, logging, is_flash_attn_2_available
+if is_flash_attn_2_available():
+    from flash_attn import flash_attn_func, flash_attn_varlen_func
+
+logger = logging.get_logger(__name__)
+
+
+class LlamaFlashAttention2(LlamaAttention):
+    """
+    Llama flash attention module. This module inherits from `LlamaAttention` as the weights of the module stays
+    untouched. The only required change would be on the forward pass where it needs to correctly call the public API of
+    flash attention and deal with padding tokens in case the input contains any of them.
+    """
+
+    def __init__(self, *args, **kwargs):
+        super().__init__(*args, **kwargs)
+
+        # TODO: Should be removed once Flash Attention for RoCm is bumped to 2.1.
+        # flash_attn<2.1 generates top-left aligned causal mask, while what is needed here is bottom-right alignement, that was made default for flash_attn>=2.1. This attribute is used to handle this difference. Reference: https://github.com/Dao-AILab/flash-attention/releases/tag/v2.1.0.
+        # Beware that with flash_attn<2.1, using q_seqlen != k_seqlen (except for the case q_seqlen == 1) produces a wrong mask (top-left).
+        self._flash_attn_uses_top_left_mask = not is_flash_attn_greater_or_equal_2_10()
+
+    def forward(
+        self,
+        hidden_states: torch.Tensor,
+        attention_mask: Optional[torch.LongTensor] = None,
+        position_ids: Optional[torch.LongTensor] = None,
+        past_key_value: Optional[Cache] = None,
+        output_attentions: bool = False,
+        use_cache: bool = False,
+        cache_position: Optional[torch.LongTensor] = None,
+        **kwargs,
+    ) -> Tuple[torch.Tensor, Optional[torch.Tensor], Optional[Tuple[torch.Tensor]]]:
+        output_attentions = False
+
+        bsz, q_len, _ = hidden_states.size()
+
+        query_states = self.q_proj(hidden_states)
+        key_states = self.k_proj(hidden_states)
+        value_states = self.v_proj(hidden_states)
+
+        # Flash attention requires the input to have the shape
+        # batch_size x seq_length x head_dim x hidden_dim
+        # therefore we just need to keep the original shape
+        query_states = query_states.view(bsz, q_len, self.num_heads, self.head_dim).transpose(1, 2)
+        key_states = key_states.view(bsz, q_len, self.num_key_value_heads, self.head_dim).transpose(1, 2)
+        value_states = value_states.view(bsz, q_len, self.num_key_value_heads, self.head_dim).transpose(1, 2)
+
+        cos, sin = self.rotary_emb(value_states, position_ids)
+
+        query_states, key_states = apply_rotary_pos_emb(query_states, key_states, cos, sin)
+
+        past_key_value = getattr(self, "past_key_value", past_key_value)
+
+        if past_key_value is not None:
+            # sin and cos are specific to RoPE models; cache_position needed for the static cache
+            cache_kwargs = {"sin": sin, "cos": cos, "cache_position": cache_position}
+            key_states, value_states = past_key_value.update(key_states, value_states, self.layer_idx, cache_kwargs)
+
+        # TODO: These transpose are quite inefficient but Flash Attention requires the layout [batch_size, sequence_length, num_heads, head_dim]. We would need to refactor the KV cache
+        # to be able to avoid many of these transpose/reshape/view.
+        query_states = query_states.transpose(1, 2)
+        key_states = key_states.transpose(1, 2)
+        value_states = value_states.transpose(1, 2)
+
+        dropout_rate = self.attention_dropout if self.training else 0.0
+
+        # In PEFT, usually we cast the layer norms in float32 for training stability reasons
+        # therefore the input hidden states gets silently casted in float32. Hence, we need
+        # cast them back in the correct dtype just to be sure everything works as expected.
+        # This might slowdown training & inference so it is recommended to not cast the LayerNorms
+        # in fp32. (LlamaRMSNorm handles it correctly)
+
+        input_dtype = query_states.dtype
+        if input_dtype == torch.float32:
+            if torch.is_autocast_enabled():
+                target_dtype = torch.get_autocast_gpu_dtype()
+            # Handle the case where the model is quantized
+            elif hasattr(self.config, "_pre_quantization_dtype"):
+                target_dtype = self.config._pre_quantization_dtype
+            else:
+                target_dtype = self.q_proj.weight.dtype
+
+            logger.warning_once(
+                f"The input hidden states seems to be silently casted in float32, this might be related to"
+                f" the fact you have upcasted embedding or layer norm layers in float32. We will cast back the input in"
+                f" {target_dtype}."
+            )
+
+            query_states = query_states.to(target_dtype)
+            key_states = key_states.to(target_dtype)
+            value_states = value_states.to(target_dtype)
+
+        attn_output = self._flash_attention_forward(
+            query_states, key_states, value_states, attention_mask, q_len, dropout=dropout_rate
+        )
+
+        attn_output = attn_output.reshape(bsz, q_len, self.hidden_size).contiguous()
+        attn_output = self.o_proj(attn_output)
+
+        if not output_attentions:
+            attn_weights = None
+
+        return attn_output, attn_weights, past_key_value
+
+    def _flash_attention_forward(
+        self, query_states, key_states, value_states, attention_mask, query_length, dropout=0.0, softmax_scale=None
+    ):
+        """
+        Calls the forward method of Flash Attention - if the input hidden states contain at least one padding token
+        first unpad the input, then computes the attention scores and pad the final attention scores.
+
+        Args:
+            query_states (`torch.Tensor`):
+                Input query states to be passed to Flash Attention API
+            key_states (`torch.Tensor`):
+                Input key states to be passed to Flash Attention API
+            value_states (`torch.Tensor`):
+                Input value states to be passed to Flash Attention API
+            attention_mask (`torch.Tensor`):
+                The padding mask - corresponds to a tensor of size `(batch_size, seq_len)` where 0 stands for the
+                position of padding tokens and 1 for the position of non-padding tokens.
+            dropout (`float`):
+                Attention dropout
+            softmax_scale (`float`, *optional*):
+                The scaling of QK^T before applying softmax. Default to 1 / sqrt(head_dim)
+        """
+        if not self._flash_attn_uses_top_left_mask:
+            causal = self.is_causal
+        else:
+            # TODO: Remove the `query_length != 1` check once Flash Attention for RoCm is bumped to 2.1. For details, please see the comment in LlamaFlashAttention2 __init__.
+            causal = self.is_causal and query_length != 1
+
+        # Contains at least one padding token in the sequence
+        if attention_mask is not None:
+            batch_size = query_states.shape[0]
+            query_states, key_states, value_states, indices_q, cu_seq_lens, max_seq_lens = self._upad_input(
+                query_states, key_states, value_states, attention_mask, query_length
+            )
+
+            cu_seqlens_q, cu_seqlens_k = cu_seq_lens
+            max_seqlen_in_batch_q, max_seqlen_in_batch_k = max_seq_lens
+
+            attn_output_unpad = flash_attn_varlen_func(
+                query_states,
+                key_states,
+                value_states,
+                cu_seqlens_q=cu_seqlens_q,
+                cu_seqlens_k=cu_seqlens_k,
+                max_seqlen_q=max_seqlen_in_batch_q,
+                max_seqlen_k=max_seqlen_in_batch_k,
+                dropout_p=dropout,
+                softmax_scale=softmax_scale,
+                causal=causal,
+            )
+
+            attn_output = pad_input(attn_output_unpad, indices_q, batch_size, query_length)
+        else:
+            attn_output = flash_attn_func(
+                query_states, key_states, value_states, dropout, softmax_scale=softmax_scale, causal=causal
+            )
+
+        return attn_output
+
+    def _upad_input(self, query_layer, key_layer, value_layer, attention_mask, query_length):
+        indices_k, cu_seqlens_k, max_seqlen_in_batch_k = _get_unpad_data(attention_mask)
+        batch_size, kv_seq_len, num_key_value_heads, head_dim = key_layer.shape
+
+        key_layer = index_first_axis(
+            key_layer.reshape(batch_size * kv_seq_len, num_key_value_heads, head_dim), indices_k
+        )
+        value_layer = index_first_axis(
+            value_layer.reshape(batch_size * kv_seq_len, num_key_value_heads, head_dim), indices_k
+        )
+        if query_length == kv_seq_len:
+            query_layer = index_first_axis(
+                query_layer.reshape(batch_size * kv_seq_len, self.num_heads, head_dim), indices_k
+            )
+            cu_seqlens_q = cu_seqlens_k
+            max_seqlen_in_batch_q = max_seqlen_in_batch_k
+            indices_q = indices_k
+        elif query_length == 1:
+            max_seqlen_in_batch_q = 1
+            cu_seqlens_q = torch.arange(
+                batch_size + 1, dtype=torch.int32, device=query_layer.device
+            )  # There is a memcpy here, that is very bad.
+            indices_q = cu_seqlens_q[:-1]
+            query_layer = query_layer.squeeze(1)
+        else:
+            # The -q_len: slice assumes left padding.
+            attention_mask = attention_mask[:, -query_length:]
+            query_layer, indices_q, cu_seqlens_q, max_seqlen_in_batch_q = unpad_input(query_layer, attention_mask)
+
+        return (
+            query_layer,
+            key_layer,
+            value_layer,
+            indices_q,
+            (cu_seqlens_q, cu_seqlens_k),
+            (max_seqlen_in_batch_q, max_seqlen_in_batch_k),
+        )
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/npu_monkey_patch/llama_rmsnorm_monkey_patch.py b/PyTorch/built-in/mlm/MiniCPM-V/npu_monkey_patch/llama_rmsnorm_monkey_patch.py
new file mode 100644
index 0000000000000000000000000000000000000000..a74bb38d0a0644163473af4db4968e5b2d3028c0
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/npu_monkey_patch/llama_rmsnorm_monkey_patch.py
@@ -0,0 +1,19 @@
+import torch
+import torch.nn as nn
+
+
+class LlamaRMSNorm(nn.Module):
+    def __init__(self, hidden_size, eps=1e-6):
+        """
+        LlamaRMSNorm is equivalent to T5LayerNorm
+        """
+        super().__init__()
+        self.weight = nn.Parameter(torch.ones(hidden_size))
+        self.variance_epsilon = eps
+
+    def forward(self, hidden_states):
+        input_dtype = hidden_states.dtype
+        hidden_states = hidden_states.to(torch.float32)
+        variance = hidden_states.pow(2).mean(-1, keepdim=True)
+        hidden_states = hidden_states * torch.rsqrt(variance + self.variance_epsilon)
+        return self.weight * hidden_states.to(input_dtype)
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/npu_monkey_patch/llama_rope_monkey_patch.py b/PyTorch/built-in/mlm/MiniCPM-V/npu_monkey_patch/llama_rope_monkey_patch.py
new file mode 100644
index 0000000000000000000000000000000000000000..3eb6cba8b0ce913851b24b38182e1f032a226950
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/npu_monkey_patch/llama_rope_monkey_patch.py
@@ -0,0 +1,120 @@
+import torch
+import torch.nn as nn
+from transformers.utils import logging
+
+logger = logging.get_logger(__name__)
+
+
+class LlamaRotaryEmbedding(nn.Module):
+    def __init__(self, dim, max_position_embeddings=2048, base=10000, device=None, scaling_factor=1.0):
+        super().__init__()
+        self.scaling_factor = scaling_factor
+        self.dim = dim
+        self.max_position_embeddings = max_position_embeddings
+        self.base = base
+        inv_freq = 1.0 / (self.base ** (torch.arange(0, self.dim, 2, dtype=torch.int64).float().to(device) / self.dim))
+        self.register_buffer("inv_freq", inv_freq, persistent=False)
+        # For BC we register cos and sin cached
+        self.max_seq_len_cached = max_position_embeddings
+        t = torch.arange(self.max_seq_len_cached, device=device, dtype=torch.int64).type_as(self.inv_freq)
+        t = t / self.scaling_factor
+        freqs = torch.outer(t, self.inv_freq)
+        # Different from paper, but it uses a different permutation in order to obtain the same calculation
+        emb = torch.cat((freqs, freqs), dim=-1)
+        self.register_buffer("_cos_cached", emb.cos().to(torch.get_default_dtype()), persistent=False)
+        self.register_buffer("_sin_cached", emb.sin().to(torch.get_default_dtype()), persistent=False)
+
+    @property
+    def sin_cached(self):
+        logger.warning_once(
+            "The sin_cached attribute will be removed in 4.39. Bear in mind that its contents changed in v4.38. Use "
+            "the forward method of RoPE from now on instead. It is not used in the `LlamaAttention` class"
+        )
+        return self._sin_cached
+
+    @property
+    def cos_cached(self):
+        logger.warning_once(
+            "The cos_cached attribute will be removed in 4.39. Bear in mind that its contents changed in v4.38. Use "
+            "the forward method of RoPE from now on instead. It is not used in the `LlamaAttention` class"
+        )
+        return self._cos_cached
+
+    @torch.no_grad()
+    def forward(self, x, position_ids):
+        # x: [bs, num_attention_heads, seq_len, head_size]
+        inv_freq_expanded = self.inv_freq[None, :, None].float().expand(position_ids.shape[0], -1, 1)
+        position_ids_expanded = position_ids[:, None, :].float()
+        # Force float32 since bfloat16 loses precision on long contexts
+        # See https://github.com/huggingface/transformers/pull/29285
+        device_type = x.device.type
+        device_type = device_type if isinstance(device_type, str) and device_type != "mps" else "cpu"
+        with torch.autocast(device_type=device_type, enabled=False):
+            freqs = (inv_freq_expanded.float() @ position_ids_expanded.float()).transpose(1, 2)
+            emb = torch.cat((freqs, freqs), dim=-1)
+            cos = emb.cos()
+            sin = emb.sin()
+        return cos.to(dtype=x.dtype), sin.to(dtype=x.dtype)
+
+
+class LlamaLinearScalingRotaryEmbedding(LlamaRotaryEmbedding):
+    """LlamaRotaryEmbedding extended with linear scaling. Credits to the Reddit user /u/kaiokendev"""
+
+    def forward(self, x, position_ids):
+        # difference to the original RoPE: a scaling factor is aplied to the position ids
+        position_ids = position_ids.float() / self.scaling_factor
+        cos, sin = super().forward(x, position_ids)
+        return cos, sin
+
+
+class LlamaDynamicNTKScalingRotaryEmbedding(LlamaRotaryEmbedding):
+    """LlamaRotaryEmbedding extended with Dynamic NTK scaling. Credits to the Reddit users /u/bloc97 and /u/emozilla"""
+
+    def forward(self, x, position_ids):
+        # difference to the original RoPE: inv_freq is recomputed when the sequence length > original length
+        seq_len = torch.max(position_ids) + 1
+        if seq_len > self.max_position_embeddings:
+            base = self.base * (
+                (self.scaling_factor * seq_len / self.max_position_embeddings) - (self.scaling_factor - 1)
+            ) ** (self.dim / (self.dim - 2))
+            inv_freq = 1.0 / (
+                base ** (torch.arange(0, self.dim, 2, dtype=torch.int64).float().to(x.device) / self.dim)
+            )
+            self.register_buffer("inv_freq", inv_freq, persistent=False)  # TODO joao: this may break with compilation
+
+        cos, sin = super().forward(x, position_ids)
+        return cos, sin
+
+
+def rotate_half(x):
+    """Rotates half the hidden dims of the input."""
+    x1 = x[..., : x.shape[-1] // 2]
+    x2 = x[..., x.shape[-1] // 2 :]
+    return torch.cat((-x2, x1), dim=-1)
+
+
+def apply_rotary_pos_emb(q, k, cos, sin, position_ids=None, unsqueeze_dim=1):
+    """Applies Rotary Position Embedding to the query and key tensors.
+
+    Args:
+        q (`torch.Tensor`): The query tensor.
+        k (`torch.Tensor`): The key tensor.
+        cos (`torch.Tensor`): The cosine part of the rotary embedding.
+        sin (`torch.Tensor`): The sine part of the rotary embedding.
+        position_ids (`torch.Tensor`, *optional*):
+            Deprecated and unused.
+        unsqueeze_dim (`int`, *optional*, defaults to 1):
+            The 'unsqueeze_dim' argument specifies the dimension along which to unsqueeze cos[position_ids] and
+            sin[position_ids] so that they can be properly broadcasted to the dimensions of q and k. For example, note
+            that cos[position_ids] and sin[position_ids] have the shape [batch_size, seq_len, head_dim]. Then, if q and
+            k have the shape [batch_size, heads, seq_len, head_dim], then setting unsqueeze_dim=1 makes
+            cos[position_ids] and sin[position_ids] broadcastable to the shapes of q and k. Similarly, if q and k have
+            the shape [batch_size, seq_len, heads, head_dim], then set unsqueeze_dim=2.
+    Returns:
+        `tuple(torch.Tensor)` comprising of the query and key tensors rotated using the Rotary Position Embedding.
+    """
+    cos = cos.unsqueeze(unsqueeze_dim)
+    sin = sin.unsqueeze(unsqueeze_dim)
+    q_embed = (q * cos) + (rotate_half(q) * sin)
+    k_embed = (k * cos) + (rotate_half(k) * sin)
+    return q_embed, k_embed
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/__init__.py b/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/constants.py b/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/constants.py
new file mode 100644
index 0000000000000000000000000000000000000000..a1ac41d363a634f45a8ba6ef681d4844a2824f5e
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/constants.py
@@ -0,0 +1,4 @@
+CONTROLLER_HEART_BEAT_EXPIRATION = 30
+WORKER_HEART_BEAT_INTERVAL = 15
+
+LOGDIR = "."
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/conversation.py b/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/conversation.py
new file mode 100644
index 0000000000000000000000000000000000000000..8e997f01f0cbe5caab5c48112189947f5392b9d5
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/conversation.py
@@ -0,0 +1,320 @@
+import dataclasses
+from enum import auto, Enum
+from typing import List, Tuple
+
+
+class SeparatorStyle(Enum):
+    """Different separator style."""
+    SINGLE = auto()
+    TWO = auto()
+
+
+@dataclasses.dataclass
+class Conversation:
+    """A class that keeps all conversation history."""
+    system: str
+    roles: List[str]
+    messages: List[List[str]]
+    offset: int
+    sep_style: SeparatorStyle = SeparatorStyle.SINGLE
+    sep: str = "###"
+    sep2: str = None
+    version: str = "Unknown"
+
+    skip_next: bool = False
+
+    def get_prompt(self):
+        if self.sep_style == SeparatorStyle.SINGLE:
+            ret = self.system + self.sep
+            for role, message in self.messages:
+                if message:
+                    if type(message) is tuple:
+                        message, _, _ = message
+                    ret += role + ": " + message + self.sep
+                else:
+                    ret += role + ":"
+            return ret
+        elif self.sep_style == SeparatorStyle.TWO:
+            seps = [self.sep, self.sep2]
+            ret = self.system + seps[0]
+            for i, (role, message) in enumerate(self.messages):
+                if message:
+                    if type(message) is tuple:
+                        message, _, _ = message
+                    ret += role + ": " + message + seps[i % 2]
+                else:
+                    ret += role + ":"
+            return ret
+        else:
+            raise ValueError(f"Invalid style: {self.sep_style}")
+
+    def append_message(self, role, message):
+        self.messages.append([role, message])
+
+    def get_images(self, return_pil=False):
+        images = []
+        for i, (role, msg) in enumerate(self.messages[self.offset:]):
+            if i % 2 == 0:
+                if type(msg) is tuple:
+                    import base64
+                    from io import BytesIO
+                    from PIL import Image
+                    msg, image, image_process_mode = msg
+                    if image_process_mode == "Pad":
+                        def expand2square(pil_img, background_color=(122, 116, 104)):
+                            width, height = pil_img.size
+                            if width == height:
+                                return pil_img
+                            elif width > height:
+                                result = Image.new(
+                                    pil_img.mode, (width, width), background_color)
+                                result.paste(
+                                    pil_img, (0, (width - height) // 2))
+                                return result
+                            else:
+                                result = Image.new(
+                                    pil_img.mode, (height, height), background_color)
+                                result.paste(
+                                    pil_img, ((height - width) // 2, 0))
+                                return result
+                        image = expand2square(image)
+                    elif image_process_mode == "Crop":
+                        pass
+                    elif image_process_mode == "Resize":
+                        image = image.resize((224, 224))
+                    else:
+                        raise ValueError(
+                            f"Invalid image_process_mode: {image_process_mode}")
+                    max_hw, min_hw = max(image.size), min(image.size)
+                    aspect_ratio = max_hw / min_hw
+                    max_len, min_len = 800, 400
+                    shortest_edge = int(
+                        min(max_len / aspect_ratio, min_len, min_hw))
+                    longest_edge = int(shortest_edge * aspect_ratio)
+                    W, H = image.size
+                    if H > W:
+                        H, W = longest_edge, shortest_edge
+                    else:
+                        H, W = shortest_edge, longest_edge
+                    image = image.resize((W, H))
+                    if return_pil:
+                        images.append(image)
+                    else:
+                        buffered = BytesIO()
+                        image.save(buffered, format="JPEG")
+                        img_b64_str = base64.b64encode(
+                            buffered.getvalue()).decode()
+                        images.append(img_b64_str)
+        return images
+
+    def to_gradio_chatbot(self):
+        ret = []
+        for i, (role, msg) in enumerate(self.messages[self.offset:]):
+            if i % 2 == 0:
+                if type(msg) is tuple:
+                    import base64
+                    from io import BytesIO
+                    msg, image, image_process_mode = msg
+                    max_hw, min_hw = max(image.size), min(image.size)
+                    aspect_ratio = max_hw / min_hw
+                    max_len, min_len = 800, 400
+                    shortest_edge = int(
+                        min(max_len / aspect_ratio, min_len, min_hw))
+                    longest_edge = int(shortest_edge * aspect_ratio)
+                    W, H = image.size
+                    if H > W:
+                        H, W = longest_edge, shortest_edge
+                    else:
+                        H, W = shortest_edge, longest_edge
+                    image = image.resize((W, H))
+                    # image = image.resize((224, 224))
+                    buffered = BytesIO()
+                    image.save(buffered, format="JPEG")
+                    img_b64_str = base64.b64encode(
+                        buffered.getvalue()).decode()
+                    img_str = f'<img src="data:image/png;base64,{img_b64_str}" alt="user upload image" />'
+                    msg = msg.replace('<image>', img_str)
+                ret.append([msg, None])
+            else:
+                ret[-1][-1] = msg
+        return ret
+
+    def copy(self):
+        return Conversation(
+            system=self.system,
+            roles=self.roles,
+            messages=[[x, y] for x, y in self.messages],
+            offset=self.offset,
+            sep_style=self.sep_style,
+            sep=self.sep,
+            sep2=self.sep2)
+
+    def dict(self):
+        if len(self.get_images()) > 0:
+            return {
+                "system": self.system,
+                "roles": self.roles,
+                "messages": [[x, y[0] if type(y) is tuple else y] for x, y in self.messages],
+                "offset": self.offset,
+                "sep": self.sep,
+                "sep2": self.sep2,
+            }
+        return {
+            "system": self.system,
+            "roles": self.roles,
+            "messages": self.messages,
+            "offset": self.offset,
+            "sep": self.sep,
+            "sep2": self.sep2,
+        }
+
+
+conv_v1 = Conversation(
+    system="A chat between a curious human and an artificial intelligence assistant. "
+           "The assistant gives helpful, detailed, and polite answers to the human's questions.",
+    roles=("Human", "Assistant"),
+    messages=(
+        ("Human", "Give three tips for staying healthy."),
+        ("Assistant",
+            "Sure, here are three tips for staying healthy:\n"
+            "1. Exercise regularly: Regular physical activity can help improve your overall health and wellbeing. "
+            "It can also help reduce your risk of chronic conditions such as obesity, diabetes, heart disease, "
+            "and certain cancers. Aim for at least 150 minutes of moderate-intensity aerobic exercise or "
+            "75 minutes of vigorous-intensity aerobic exercise per week, along with muscle-strengthening "
+            "activities at least two days per week.\n"
+            "2. Eat a balanced diet: Eating a balanced diet that is rich in fruits, "
+            "vegetables, whole grains, lean proteins, and healthy fats can help support "
+            "your overall health. Try to limit your intake of processed and high-sugar foods, "
+            "and aim to drink plenty of water throughout the day.\n"
+            "3. Get enough sleep: Getting enough quality sleep is essential for your physical "
+            "and mental health. Adults should aim for seven to nine hours of sleep per night. "
+            "Establish a regular sleep schedule and try to create a relaxing bedtime routine to "
+            "help improve the quality of your sleep.")
+    ),
+    offset=2,
+    sep_style=SeparatorStyle.SINGLE,
+    sep="###",
+)
+
+conv_v1_2 = Conversation(
+    system="A chat between a curious human and an artificial intelligence assistant. "
+           "The assistant gives helpful, detailed, and polite answers to the human's questions.",
+    roles=("Human", "Assistant"),
+    messages=(
+        ("Human", "What are the key differences between renewable and non-renewable energy sources?"),
+        ("Assistant",
+            "Renewable energy sources are those that can be replenished naturally in a relatively "
+            "short amount of time, such as solar, wind, hydro, geothermal, and biomass. "
+            "Non-renewable energy sources, on the other hand, are finite and will eventually be "
+            "depleted, such as coal, oil, and natural gas. Here are some key differences between "
+            "renewable and non-renewable energy sources:\n"
+            "1. Availability: Renewable energy sources are virtually inexhaustible, while non-renewable "
+            "energy sources are finite and will eventually run out.\n"
+            "2. Environmental impact: Renewable energy sources have a much lower environmental impact "
+            "than non-renewable sources, which can lead to air and water pollution, greenhouse gas emissions, "
+            "and other negative effects.\n"
+            "3. Cost: Renewable energy sources can be more expensive to initially set up, but they typically "
+            "have lower operational costs than non-renewable sources.\n"
+            "4. Reliability: Renewable energy sources are often more reliable and can be used in more remote "
+            "locations than non-renewable sources.\n"
+            "5. Flexibility: Renewable energy sources are often more flexible and can be adapted to different "
+            "situations and needs, while non-renewable sources are more rigid and inflexible.\n"
+            "6. Sustainability: Renewable energy sources are more sustainable over the long term, while "
+            "non-renewable sources are not, and their depletion can lead to economic and social instability.\n")
+    ),
+    offset=2,
+    sep_style=SeparatorStyle.SINGLE,
+    sep="###",
+)
+
+conv_vicuna_v1_1 = Conversation(
+    system="A chat between a curious user and an artificial intelligence assistant. "
+    "The assistant gives helpful, detailed, and polite answers to the user's questions.",
+    roles=("USER", "ASSISTANT"),
+    version="v1",
+    messages=(),
+    offset=0,
+    sep_style=SeparatorStyle.TWO,
+    sep=" ",
+    sep2="</s>",
+)
+
+conv_bair_v1 = Conversation(
+    system="BEGINNING OF CONVERSATION:",
+    roles=("USER", "GPT"),
+    messages=(),
+    offset=0,
+    sep_style=SeparatorStyle.TWO,
+    sep=" ",
+    sep2="</s>",
+)
+
+simple_conv = Conversation(
+    system="You are LLaVA, a large language model trained by UW Madison WAIV Lab, based on LLaMA architecture."
+           "You are designed to assist human with a variety of tasks using natural language."
+           "Follow the instructions carefully.",
+    roles=("Human", "Assistant"),
+    messages=(
+        ("Human", "Hi!"),
+        ("Assistant", "Hi there!  How can I help you today?\n")
+    ),
+    offset=2,
+    sep_style=SeparatorStyle.SINGLE,
+    sep="###",
+)
+
+simple_conv_multimodal = Conversation(
+    system="A chat between a curious user and an artificial intelligence assistant. "
+    "The assistant gives helpful, detailed, and polite answers to the user's questions.",
+    roles=("Human", "Assistant"),
+    messages=(
+    ),
+    offset=0,
+    sep_style=SeparatorStyle.SINGLE,
+    sep="###",
+)
+
+simple_conv_legacy = Conversation(
+    system="You are LLaVA, a large language model trained by UW Madison WAIV Lab."
+           "You are designed to assist human with a variety of tasks using natural language."
+           "Follow the instructions carefully.",
+    roles=("Human", "Assistant"),
+    messages=(
+        ("Human", "Hi!\n\n### Response:"),
+        ("Assistant", "Hi there!  How can I help you today?\n")
+    ),
+    offset=2,
+    sep_style=SeparatorStyle.SINGLE,
+    sep="###",
+)
+
+conv_llava_v1 = Conversation(
+    system="You are LLaVA, a large language and vision assistant trained by UW Madison WAIV Lab."
+           "You are able to understand the visual content that the user provides, and assist the user with a variety of tasks using natural language."
+           "Follow the instructions carefully and explain your answers in detail.",
+    roles=("USER", "ASSISTANT"),
+    version="v1",
+    messages=(),
+    offset=0,
+    sep_style=SeparatorStyle.TWO,
+    sep=" ",
+    sep2="</s>",
+)
+
+default_conversation = conv_v1_2
+conv_templates = {
+    "default": conv_v1_2,
+    "simple": simple_conv,
+    "simple_legacy": simple_conv_legacy,
+    "multimodal": simple_conv_multimodal,
+    "llava_v1": conv_llava_v1,
+
+    # fastchat
+    "v1": conv_v1_2,
+    "bair_v1": conv_bair_v1,
+    "vicuna_v1_1": conv_vicuna_v1_1,
+}
+
+
+if __name__ == "__main__":
+    print(default_conversation.get_prompt())
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/model/__init__.py b/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/model/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..35b9845c8726de2a6b82fb05ab11071115f6b09c
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/model/__init__.py
@@ -0,0 +1 @@
+from .omnilmm import OmniLMMForCausalLM
\ No newline at end of file
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/model/omnilmm.py b/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/model/omnilmm.py
new file mode 100644
index 0000000000000000000000000000000000000000..e720675292e9d8e14e25a057624ae5c290f82d5d
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/model/omnilmm.py
@@ -0,0 +1,457 @@
+
+import gc
+import math
+import timm
+import torch
+from torch import Tensor
+import torch.nn as nn
+from torch.nn import CrossEntropyLoss
+from typing import List, Optional, Tuple, Union
+
+from transformers import AutoConfig, AutoModelForCausalLM
+from transformers import MistralForCausalLM, MistralModel, MistralConfig
+from transformers.modeling_outputs import BaseModelOutputWithPast, CausalLMOutputWithPast
+
+from omnilmm.model.utils import build_transform
+from omnilmm.model.resampler import Resampler
+
+DEFAULT_IMAGE_PATCH_TOKEN = "<im_patch>"
+DEFAULT_IM_START_TOKEN = "<im_start>"
+DEFAULT_IM_END_TOKEN = "<im_end>"
+
+
+class OmniLMMConfig(MistralConfig):
+    model_type = "omnilmm"
+
+
+class Identity(torch.nn.Identity):
+    def forward(self, input: Tensor, **kwargs) -> Tensor:
+        return super().forward(input)
+
+
+def create_vision_module(config):
+    vision_tower = timm.create_model('eva02_enormous_patch14_clip_224.laion2b_plus',
+                                     pretrained=False,
+                                     num_classes=0,
+                                     dynamic_img_size=True,
+                                     dynamic_img_pad=True)
+
+    if isinstance(vision_tower, timm.models.VisionTransformer):
+        if vision_tower.attn_pool is not None:
+            vision_tower.attn_pool = Identity()
+
+    # use 2nd last layer's output
+    vision_tower.blocks[-1] = Identity()
+
+    embed_dim = config.hidden_size
+    resampler = Resampler(
+        grid_size=int(math.sqrt(config.num_query)),
+        embed_dim=embed_dim,
+        num_heads=embed_dim // 128,
+        kv_dim=vision_tower.embed_dim,
+    )
+    return vision_tower, resampler
+
+
+class OmniLMMModel(MistralModel):
+    config_class = OmniLMMConfig
+
+    def __init__(self, config: OmniLMMConfig, mm_vision_tower=None, mm_hidden_size=None, tune_clip=True):
+        super(OmniLMMModel, self).__init__(config)
+
+        if hasattr(config, "mm_vision_tower"):
+            vision_tower, resampler = create_vision_module(config)
+
+            # print(__file__, 'skip loading vision tower weights')
+
+            # HACK: for FSDP
+            self.vision_tower = [vision_tower]
+            self.resampler = resampler
+            if tune_clip:
+                self.vision_tower = self.vision_tower[0]
+
+        self.vision_config = lambda x: None
+
+    def initialize_vision_modules(self, vision_tower, no_randaug, num_query, image_size, tune_clip=False):
+        self.config.mm_vision_tower = vision_tower
+        self.config.use_mm_proj = True
+        self.config.num_query = num_query
+        self.config.image_size = image_size
+
+        if not hasattr(self, 'vision_tower'):
+            vision_tower, resampler = create_vision_module(self.config)
+            state_dict = torch.load(
+                '/tt/data/public/multimodal/multimodal_model_ckpts/timm/eva02_enormous_patch14_clip_224.laion2b_plus.pt')
+            vision_tower.load_state_dict(state_dict, strict=False)
+            del state_dict
+            gc.collect()
+        else:
+            if isinstance(self.vision_tower, list):
+                vision_tower = self.vision_tower[0]
+            else:
+                vision_tower = self.vision_tower
+            resampler = self.resampler
+        self.vision_tower = vision_tower if tune_clip else [vision_tower]
+        self.resampler = resampler
+
+        train_img_transform = build_transform(
+            is_train=True, randaug=not no_randaug, input_size=self.config.image_size, std_mode='OPENAI_CLIP')
+        eval_img_transform = build_transform(
+            is_train=False, input_size=self.config.image_size, std_mode='OPENAI_CLIP')
+
+        return dict(
+            image_processor=(train_img_transform, eval_img_transform),
+            image_token_len=num_query,
+            vision_config=self.vision_config
+        )
+
+    def get_vision_embedding(self, pixel_values):
+        if isinstance(self.vision_tower, list):
+            vision_tower = self.vision_tower[0]  # HACK: for FSDP
+        else:
+            vision_tower = self.vision_tower
+
+        dtype = vision_tower.pos_embed.data.dtype
+        vision_embedding = vision_tower.forward_features(
+            pixel_values.type(dtype))
+        if hasattr(vision_tower, 'num_prefix_tokens') and vision_tower.num_prefix_tokens > 0:
+            vision_embedding = vision_embedding[:,
+                                                vision_tower.num_prefix_tokens:]
+        res = self.resampler(vision_embedding)
+        return res
+
+    def get_vllm_embedding(self, data):
+
+        if 'vision_hidden_states' not in data:
+            pixel_values_list = data['pixel_values']
+            vision_hidden_states = []
+            for pixel_values in pixel_values_list:
+                if len(pixel_values) > 0:
+                    vision_hidden_states.append(self.get_vision_embedding(pixel_values.unsqueeze(0))[0])
+                else:
+                    vision_hidden_states.append([])
+        else:
+            vision_hidden_states = data['vision_hidden_states']
+
+        #vllm_embedding = self.llm.model.embed_tokens(data['input_ids']) * self.llm.config.scale_emb
+        inputs_embeds = self.embed_tokens(data['input_ids'])
+        vision_hidden_states = [i.type(inputs_embeds.dtype) 
+            if isinstance(i, torch.Tensor) else i for i in vision_hidden_states
+        ]
+
+
+        # HACK: replace back original embeddings for LLaVA pretraining
+        orig_embeds_params = getattr(self, 'orig_embeds_params', None)
+
+        new_input_embeds = []
+        cur_image_idx = 0
+        for cur_input_ids, cur_input_embeds in zip(data['input_ids'], inputs_embeds):
+            if (cur_input_ids == self.vision_config.im_patch_token).sum() == 0:
+                # multimodal LLM, but the current sample is not multimodal
+                cur_input_embeds = cur_input_embeds + (0. * dummy_image_features).sum()
+                new_input_embeds.append(cur_input_embeds)
+                continue
+
+            if self.vision_config.use_im_start_end:
+                cur_image_features = vision_hidden_states[cur_image_idx]
+                num_patches = cur_image_features.shape[0]
+                if (cur_input_ids == self.vision_config.im_start_token).sum() != (cur_input_ids == self.vision_config.im_end_token).sum():
+                    raise ValueError(
+                        "The number of image start tokens and image end tokens should be the same.")
+                image_start_tokens = torch.where(
+                    cur_input_ids == self.vision_config.im_start_token)[0]
+                for image_start_token_pos in image_start_tokens:
+                    cur_image_features = vision_hidden_states[cur_image_idx].to(
+                        device=cur_input_embeds.device)
+                    num_patches = cur_image_features.shape[0]
+                    if cur_input_ids[image_start_token_pos + num_patches + 1] != self.vision_config.im_end_token:
+                        raise ValueError(
+                            "The image end token should follow the image start token.")
+                    if orig_embeds_params is not None:
+                        cur_new_input_embeds = torch.cat((cur_input_embeds[:image_start_token_pos].detach(), cur_input_embeds[image_start_token_pos:image_start_token_pos+1], cur_image_features,
+                                                         cur_input_embeds[image_start_token_pos + num_patches + 1:image_start_token_pos + num_patches + 2], cur_input_embeds[image_start_token_pos + num_patches + 2:].detach()), dim=0)
+                    else:
+                        cur_new_input_embeds = torch.cat(
+                            (cur_input_embeds[:image_start_token_pos+1], cur_image_features, cur_input_embeds[image_start_token_pos + num_patches + 1:]), dim=0)
+                    cur_image_idx += 1
+                new_input_embeds.append(cur_new_input_embeds)
+            else:
+                raise NotImplementedError
+        inputs_embeds = torch.stack(new_input_embeds, dim=0)
+
+        return inputs_embeds, vision_hidden_states
+
+    def forward(
+        self,
+        input_ids: torch.LongTensor = None,
+        attention_mask: Optional[torch.Tensor] = None,
+        past_key_values: Optional[List[torch.FloatTensor]] = None,
+        inputs_embeds: Optional[torch.FloatTensor] = None,
+        use_cache: Optional[bool] = None,
+        output_attentions: Optional[bool] = None,
+        output_hidden_states: Optional[bool] = None,
+        images: Optional[torch.FloatTensor] = None,
+        return_dict: Optional[bool] = None,
+        **kwargs
+    ) -> Union[Tuple, BaseModelOutputWithPast]:
+
+        # HACK: replace back original embeddings for LLaVA pretraining
+        orig_embeds_params = getattr(self, 'orig_embeds_params', None)
+
+        if inputs_embeds is None and past_key_values is None:
+          inputs_embeds = self.embed_tokens(input_ids)
+
+          vision_tower = getattr(self, 'vision_tower', None)
+          if vision_tower is not None and (input_ids.shape[1] != 1 or self.training) and images is not None:
+
+            if type(images) is list:
+                image_features = []
+                for image in images:
+                    image_forward_out = self.get_vision_embedding(image.unsqueeze(0))[
+                        0]
+                    image_features.append(image_forward_out)
+            else:
+                image_features = self.get_vision_embedding(images)
+
+            dummy_image_features = torch.zeros(
+                self.config.num_query,
+                self.config.hidden_size,
+                device=inputs_embeds.device,
+                dtype=inputs_embeds.dtype)
+
+            new_input_embeds = []
+            cur_image_idx = 0
+            for cur_input_ids, cur_input_embeds in zip(input_ids, inputs_embeds):
+                if (cur_input_ids == self.vision_config.im_patch_token).sum() == 0:
+                    # multimodal LLM, but the current sample is not multimodal
+                    cur_input_embeds = cur_input_embeds + \
+                        (0. * dummy_image_features).sum()
+                    new_input_embeds.append(cur_input_embeds)
+                    continue
+
+                if self.vision_config.use_im_start_end:
+                    cur_image_features = image_features[cur_image_idx]
+                    num_patches = cur_image_features.shape[0]
+                    if (cur_input_ids == self.vision_config.im_start_token).sum() != (cur_input_ids == self.vision_config.im_end_token).sum():
+                        raise ValueError(
+                            "The number of image start tokens and image end tokens should be the same.")
+                    image_start_tokens = torch.where(
+                        cur_input_ids == self.vision_config.im_start_token)[0]
+                    for image_start_token_pos in image_start_tokens:
+                        cur_image_features = image_features[cur_image_idx].to(
+                            device=cur_input_embeds.device)
+                        num_patches = cur_image_features.shape[0]
+                        if cur_input_ids[image_start_token_pos + num_patches + 1] != self.vision_config.im_end_token:
+                            raise ValueError(
+                                "The image end token should follow the image start token.")
+                        if orig_embeds_params is not None:
+                            cur_new_input_embeds = torch.cat((cur_input_embeds[:image_start_token_pos].detach(), cur_input_embeds[image_start_token_pos:image_start_token_pos+1], cur_image_features,
+                                                             cur_input_embeds[image_start_token_pos + num_patches + 1:image_start_token_pos + num_patches + 2], cur_input_embeds[image_start_token_pos + num_patches + 2:].detach()), dim=0)
+                        else:
+                            cur_new_input_embeds = torch.cat(
+                                (cur_input_embeds[:image_start_token_pos+1], cur_image_features, cur_input_embeds[image_start_token_pos + num_patches + 1:]), dim=0)
+                        cur_image_idx += 1
+                    new_input_embeds.append(cur_new_input_embeds)
+                else:
+                    raise NotImplementedError
+            inputs_embeds = torch.stack(new_input_embeds, dim=0)
+            input_ids = None
+
+        return super(OmniLMMModel, self).forward(
+            input_ids=input_ids, attention_mask=attention_mask, past_key_values=past_key_values,
+            inputs_embeds=inputs_embeds, use_cache=use_cache,
+            output_attentions=output_attentions, output_hidden_states=output_hidden_states,
+            return_dict=return_dict,
+            **kwargs
+        )
+
+
+class OmniLMMForCausalLM(MistralForCausalLM):
+    config_class = OmniLMMConfig
+
+    def __init__(self, config, mm_vision_tower=None, tune_clip=True):
+        super(MistralForCausalLM, self).__init__(config)
+        self.model = OmniLMMModel(
+            config, mm_vision_tower=mm_vision_tower, tune_clip=tune_clip)
+
+        self.lm_head = nn.Linear(
+            config.hidden_size, config.vocab_size, bias=False)
+
+        # Initialize weights and apply final processing
+        self.post_init()
+
+    def forward(
+        self,
+        input_ids: torch.LongTensor = None,
+        attention_mask: Optional[torch.Tensor] = None,
+        past_key_values: Optional[List[torch.FloatTensor]] = None,
+        inputs_embeds: Optional[torch.FloatTensor] = None,
+        labels: Optional[torch.LongTensor] = None,
+        use_cache: Optional[bool] = None,
+        output_attentions: Optional[bool] = None,
+        output_hidden_states: Optional[bool] = None,
+        images: Optional[torch.FloatTensor] = None,
+        return_dict: Optional[bool] = None,
+        **kwargs
+    ) -> Union[Tuple, CausalLMOutputWithPast]:
+        output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
+        output_hidden_states = (
+            output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
+        )
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+
+        # print(f'@@@ At forward, labels: {labels.shape}-{labels}', flush=True)
+        # print(f'@@@ At forward, input_ids: {input_ids.shape}-{input_ids}', flush=True)
+        # print(f'@@@ At forward, input_ids: {attention_mask.shape}-{attention_mask}', flush=True)
+
+        # decoder outputs consists of (dec_features, layer_state, dec_hidden, dec_attn)
+        outputs = self.model(
+            input_ids=input_ids,
+            attention_mask=attention_mask,
+            past_key_values=past_key_values,
+            inputs_embeds=inputs_embeds,
+            use_cache=use_cache,
+            output_attentions=output_attentions,
+            output_hidden_states=output_hidden_states,
+            return_dict=return_dict,
+            images=images,
+            **kwargs
+        )
+
+        hidden_states = outputs[0]
+        logits = self.lm_head(hidden_states)
+
+        loss = None
+        if labels is not None:
+            # Shift so that tokens < n predict n
+            shift_logits = logits[..., :-1, :].contiguous()
+            shift_labels = labels[..., 1:].contiguous()
+            # Flatten the tokens
+            loss_fct = CrossEntropyLoss()
+            shift_logits = shift_logits.view(-1, self.config.vocab_size)
+            shift_labels = shift_labels.view(-1)
+            # Enable model/pipeline parallelism
+            shift_labels = shift_labels.to(shift_logits.device)
+            loss = loss_fct(shift_logits, shift_labels)
+
+        if not return_dict:
+            output = (logits,) + outputs[1:]
+            return (loss,) + output if loss is not None else output
+
+        return CausalLMOutputWithPast(
+            loss=loss,
+            logits=logits,
+            past_key_values=outputs.past_key_values,
+            hidden_states=outputs.hidden_states,
+            attentions=outputs.attentions,
+        )
+
+    # TODO could be removed for generate_vllm()
+    def prepare_inputs_for_generation(
+        self, input_ids, past_key_values=None, attention_mask=None, inputs_embeds=None, **kwargs
+    ):
+        if past_key_values:
+            input_ids = input_ids[:, -1:]
+
+        # if `inputs_embeds` are passed, we only want to use them in the 1st generation step
+        if inputs_embeds is not None and past_key_values is None:
+            model_inputs = {"inputs_embeds": inputs_embeds}
+        else:
+            model_inputs = {"input_ids": input_ids}
+
+        model_inputs.update(
+            {
+                "past_key_values": past_key_values,
+                "use_cache": kwargs.get("use_cache"),
+                "attention_mask": attention_mask,
+                "images": kwargs.get("images", None),
+            }
+        )
+        return model_inputs
+
+    def generate_vllm(
+        self,
+        input_ids: torch.LongTensor = None,
+        images: Optional[torch.FloatTensor] = None,
+        vision_hidden_states=None,
+        return_vision_hidden_states=False,
+        **kwargs
+    ):
+        model_inputs = {'input_ids': input_ids}
+        if vision_hidden_states is None:
+            model_inputs['pixel_values'] = images
+        else:
+            model_inputs['vision_hidden_states'] = vision_hidden_states
+
+        with torch.inference_mode():
+            inputs_embeds, vision_hidden_states = self.model.get_vllm_embedding(model_inputs)
+
+            result = self.generate(
+                inputs_embeds=inputs_embeds,
+                **kwargs
+            )
+
+        if return_vision_hidden_states:
+            return result, vision_hidden_states
+
+        return result
+
+
+    def initialize_vision_tokenizer(self, mm_use_im_start_end, tokenizer, device,
+                                    tune_mm_mlp_adapter=False):
+        self.model.vision_config.use_im_start_end = mm_use_im_start_end
+        tokenizer.add_tokens([DEFAULT_IMAGE_PATCH_TOKEN], special_tokens=True)
+        self.resize_token_embeddings(len(tokenizer))
+
+        if mm_use_im_start_end:
+            num_new_tokens = tokenizer.add_tokens(
+                [DEFAULT_IM_START_TOKEN, DEFAULT_IM_END_TOKEN], special_tokens=True)
+            self.resize_token_embeddings(len(tokenizer))
+            self.model.vision_config.im_start_token, self.model.vision_config.im_end_token = tokenizer.convert_tokens_to_ids(
+                [DEFAULT_IM_START_TOKEN, DEFAULT_IM_END_TOKEN])
+
+            if num_new_tokens > 0:
+                input_embeddings = self.get_input_embeddings().weight.data
+                output_embeddings = self.get_output_embeddings().weight.data
+
+                input_embeddings_avg = input_embeddings[:-num_new_tokens].mean(
+                    dim=0, keepdim=True)
+                output_embeddings_avg = output_embeddings[:-num_new_tokens].mean(
+                    dim=0, keepdim=True)
+
+                input_embeddings[-num_new_tokens:] = input_embeddings_avg
+                output_embeddings[-num_new_tokens:] = output_embeddings_avg
+
+            # for new sft data
+            num_new_tokens = tokenizer.add_tokens(
+                ['<box>', '</box>', '<ref>', '</ref>', '<quad>', '</quad>'], special_tokens=True)
+            self.resize_token_embeddings(len(tokenizer))
+
+            if num_new_tokens > 0:
+                input_embeddings = self.get_input_embeddings().weight.data
+                output_embeddings = self.get_output_embeddings().weight.data
+
+                input_embeddings_avg = input_embeddings[:-num_new_tokens].mean(
+                    dim=0, keepdim=True)
+                output_embeddings_avg = output_embeddings[:-num_new_tokens].mean(
+                    dim=0, keepdim=True)
+
+                input_embeddings[-num_new_tokens:] = input_embeddings_avg
+                output_embeddings[-num_new_tokens:] = output_embeddings_avg
+
+            if tune_mm_mlp_adapter:
+                self.model.orig_embeds_params = [
+                    self.get_input_embeddings().weight.data.clone().to(device=device)]
+                for p in self.get_input_embeddings().parameters():
+                    p.requires_grad = True
+                for p in self.get_output_embeddings().parameters():
+                    p.requires_grad = False
+
+        self.model.vision_config.im_patch_token = tokenizer.convert_tokens_to_ids(
+            [DEFAULT_IMAGE_PATCH_TOKEN])[0]
+        print(f'Tokenizer: {tokenizer}\n patch_token_id: {self.model.vision_config.im_patch_token}, visoin_config: {self.model.vision_config}', flush=True)
+        # exit()
+
+
+AutoConfig.register("omnilmm", OmniLMMConfig)
+AutoModelForCausalLM.register(OmniLMMConfig, OmniLMMForCausalLM)
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/model/resampler.py b/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/model/resampler.py
new file mode 100644
index 0000000000000000000000000000000000000000..0d46f55f460384ab23d0b4ff00768417281110dc
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/model/resampler.py
@@ -0,0 +1,171 @@
+# Copyright (c) Alibaba Cloud.
+#
+# This source code is licensed under the license found in the
+# LICENSE file in the root directory of this source tree.
+
+from collections import OrderedDict
+import math
+import requests
+from io import BytesIO
+from functools import partial
+from PIL import Image
+from typing import Callable, Optional, Sequence, Tuple, List, Union
+import numpy as np
+
+import torch
+from torch import nn
+from torch.nn import functional as F
+from torch.nn.init import trunc_normal_
+from torchvision import transforms
+from torchvision.transforms import InterpolationMode
+
+
+def get_abs_pos(abs_pos, tgt_size):
+    # abs_pos: L, C
+    # tgt_size: M
+    # return: M, C
+    src_size = int(math.sqrt(abs_pos.size(0)))
+    tgt_size = int(math.sqrt(tgt_size))
+    dtype = abs_pos.dtype
+
+    if src_size != tgt_size:
+        return F.interpolate(
+            abs_pos.float().reshape(1, src_size, src_size, -1).permute(0, 3, 1, 2),
+            size=(tgt_size, tgt_size),
+            mode="bicubic",
+            align_corners=False,
+        ).permute(0, 2, 3, 1).flatten(0, 2).to(dtype=dtype)
+    else:
+        return abs_pos
+
+
+# https://github.com/facebookresearch/mae/blob/efb2a8062c206524e35e47d04501ed4f544c0ae8/util/pos_embed.py#L20
+def get_2d_sincos_pos_embed(embed_dim, grid_size, cls_token=False):
+    """
+    grid_size: int of the grid height and width
+    return:
+    pos_embed: [grid_size*grid_size, embed_dim] or [1+grid_size*grid_size, embed_dim] (w/ or w/o cls_token)
+    """
+    grid_h = np.arange(grid_size, dtype=np.float32)
+    grid_w = np.arange(grid_size, dtype=np.float32)
+    grid = np.meshgrid(grid_w, grid_h)  # here w goes first
+    grid = np.stack(grid, axis=0)
+
+    grid = grid.reshape([2, 1, grid_size, grid_size])
+    pos_embed = get_2d_sincos_pos_embed_from_grid(embed_dim, grid)
+    if cls_token:
+        pos_embed = np.concatenate(
+            [np.zeros([1, embed_dim]), pos_embed], axis=0)
+    return pos_embed
+
+
+def get_2d_sincos_pos_embed_from_grid(embed_dim, grid):
+    assert embed_dim % 2 == 0
+
+    # use half of dimensions to encode grid_h
+    emb_h = get_1d_sincos_pos_embed_from_grid(
+        embed_dim // 2, grid[0])  # (H*W, D/2)
+    emb_w = get_1d_sincos_pos_embed_from_grid(
+        embed_dim // 2, grid[1])  # (H*W, D/2)
+
+    emb = np.concatenate([emb_h, emb_w], axis=1)  # (H*W, D)
+    return emb
+
+
+def get_1d_sincos_pos_embed_from_grid(embed_dim, pos):
+    """
+    embed_dim: output dimension for each position
+    pos: a list of positions to be encoded: size (M,)
+    out: (M, D)
+    """
+    assert embed_dim % 2 == 0
+    omega = np.arange(embed_dim // 2, dtype=np.float32)
+    omega /= embed_dim / 2.
+    omega = 1. / 10000 ** omega  # (D/2,)
+
+    pos = pos.reshape(-1)  # (M,)
+    out = np.einsum('m,d->md', pos, omega)  # (M, D/2), outer product
+
+    emb_sin = np.sin(out)  # (M, D/2)
+    emb_cos = np.cos(out)  # (M, D/2)
+
+    emb = np.concatenate([emb_sin, emb_cos], axis=1)  # (M, D)
+    return emb
+
+
+class Resampler(nn.Module):
+    """
+    A 2D perceiver-resampler network with one cross attention layers by
+        (grid_size**2) learnable queries and 2d sincos pos_emb
+    Outputs:
+        A tensor with the shape of (grid_size**2, embed_dim)
+    """
+
+    def __init__(
+            self,
+            grid_size,
+            embed_dim,
+            num_heads,
+            kv_dim=None,
+            norm_layer=partial(nn.LayerNorm, eps=1e-6)
+    ):
+        super().__init__()
+        self.num_queries = grid_size ** 2
+        self.embed_dim = embed_dim
+        self.num_heads = num_heads
+
+        self.pos_embed = nn.Parameter(
+            torch.from_numpy(get_2d_sincos_pos_embed(
+                embed_dim, grid_size)).float()
+        ).requires_grad_(False)
+
+        self.query = nn.Parameter(torch.zeros(self.num_queries, embed_dim))
+        trunc_normal_(self.query, std=.02)
+
+        if kv_dim is not None and kv_dim != embed_dim:
+            self.kv_proj = nn.Linear(kv_dim, embed_dim, bias=False)
+        else:
+            self.kv_proj = nn.Identity()
+
+        self.attn = nn.MultiheadAttention(embed_dim, num_heads)
+        self.ln_q = norm_layer(embed_dim)
+        self.ln_kv = norm_layer(embed_dim)
+
+        self.ln_post = norm_layer(embed_dim)
+        self.proj = nn.Parameter(
+            (embed_dim ** -0.5) * torch.randn(embed_dim, embed_dim))
+
+        self.apply(self._init_weights)
+
+    def _init_weights(self, m):
+        if isinstance(m, nn.Linear):
+            trunc_normal_(m.weight, std=.02)
+            if isinstance(m, nn.Linear) and m.bias is not None:
+                nn.init.constant_(m.bias, 0)
+        elif isinstance(m, nn.LayerNorm):
+            nn.init.constant_(m.bias, 0)
+            nn.init.constant_(m.weight, 1.0)
+
+    def forward(self, x, attn_mask=None):
+
+        pos_embed = get_abs_pos(self.pos_embed, x.size(1))
+
+        x = self.kv_proj(x)
+        x = self.ln_kv(x).permute(1, 0, 2)
+
+        N = x.shape[1]
+        q = self.ln_q(self.query)
+        # print((self._repeat(q, N) + self.pos_embed.unsqueeze(1)).dtype, (x + pos_embed.unsqueeze(1)).dtype, x.dtype)
+        out = self.attn(
+            self._repeat(q, N) + self.pos_embed.unsqueeze(1),
+            x + pos_embed.unsqueeze(1),
+            x,
+            attn_mask=attn_mask)[0]
+        x = out.permute(1, 0, 2)
+
+        x = self.ln_post(x)
+        x = x @ self.proj
+        return x
+
+    def _repeat(self, query, N: int):
+        return query.unsqueeze(1).repeat(1, N, 1)
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/model/utils.py b/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/model/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..fcb8eba1f74b111aa68be3aa0211c831cf243cb9
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/model/utils.py
@@ -0,0 +1,555 @@
+from torchvision import transforms
+from timm.data.transforms import RandomResizedCropAndInterpolation
+from timm.data.constants import IMAGENET_INCEPTION_MEAN, IMAGENET_INCEPTION_STD
+from transformers import AutoConfig
+from PIL import Image
+from io import BytesIO
+import torch.distributed as dist
+import numpy as np
+import pickle
+import base64
+import cv2
+import os
+import torch
+from transformers import AutoConfig, StoppingCriteria
+
+try:
+    from timm.data.constants import OPENAI_CLIP_MEAN, OPENAI_CLIP_STD
+except ImportError:
+    OPENAI_CLIP_MEAN = (0.48145466, 0.4578275, 0.40821073)
+    OPENAI_CLIP_STD = (0.26862954, 0.26130258, 0.27577711)
+
+
+def auto_upgrade(config):
+    cfg = AutoConfig.from_pretrained(config)
+    if 'llava' in config and cfg.model_type != 'llava':
+        print("You are using newer LLaVA code base, while the checkpoint of v0 is from older code base.")
+        print("You must upgrade the checkpoint to the new code base (this can be done automatically).")
+        confirm = input(
+            "Please confirm that you want to upgrade the checkpoint. [Y/N]")
+        if confirm.lower() in ["y", "yes"]:
+            print("Upgrading checkpoint...")
+            assert len(cfg.architectures) == 1
+            setattr(cfg.__class__, "model_type", "llava")
+            cfg.architectures[0] = 'LlavaLlamaForCausalLM'
+            cfg.save_pretrained(config)
+            print("Checkpoint upgraded.")
+        else:
+            print("Checkpoint upgrade aborted.")
+            exit(1)
+
+
+class KeywordsStoppingCriteria(StoppingCriteria):
+    def __init__(self, keywords, tokenizer, input_ids):
+        self.keywords = keywords
+        self.tokenizer = tokenizer
+        self.start_len = None
+        self.input_ids = input_ids
+
+    def __call__(self, output_ids: torch.LongTensor, scores: torch.FloatTensor, **kwargs) -> bool:
+        if self.start_len is None:
+            self.start_len = self.input_ids.shape[1]
+        else:
+            outputs = self.tokenizer.batch_decode(
+                output_ids[:, self.start_len:], skip_special_tokens=True)[0]
+            for keyword in self.keywords:
+                if keyword in outputs:
+                    return True
+        return False
+
+
+def auto_upgrade(config):
+    cfg = AutoConfig.from_pretrained(config)
+    if 'llava' in config and cfg.model_type != 'llava':
+        print("You are using newer LLaVA code base, while the checkpoint of v0 is from older code base.")
+        print("You must upgrade the checkpoint to the new code base (this can be done automatically).")
+        confirm = input(
+            "Please confirm that you want to upgrade the checkpoint. [Y/N]")
+        if confirm.lower() in ["y", "yes"]:
+            print("Upgrading checkpoint...")
+            assert len(cfg.architectures) == 1
+            setattr(cfg.__class__, "model_type", "llava")
+            cfg.architectures[0] = 'LlavaLlamaForCausalLM'
+            cfg.save_pretrained(config)
+            print("Checkpoint upgraded.")
+        else:
+            print("Checkpoint upgrade aborted.")
+            exit(1)
+
+# aug functions
+
+
+def identity_func(img):
+    return img
+
+
+def autocontrast_func(img, cutoff=0):
+    '''
+        same output as PIL.ImageOps.autocontrast
+    '''
+    n_bins = 256
+
+    def tune_channel(ch):
+        n = ch.size
+        cut = cutoff * n // 100
+        if cut == 0:
+            high, low = ch.max(), ch.min()
+        else:
+            hist = cv2.calcHist([ch], [0], None, [n_bins], [0, n_bins])
+            low = np.argwhere(np.cumsum(hist) > cut)
+            low = 0 if low.shape[0] == 0 else low[0]
+            high = np.argwhere(np.cumsum(hist[::-1]) > cut)
+            high = n_bins - 1 if high.shape[0] == 0 else n_bins - 1 - high[0]
+        if high <= low:
+            table = np.arange(n_bins)
+        else:
+            scale = (n_bins - 1) / (high - low)
+            table = np.arange(n_bins) * scale - low * scale
+            table[table < 0] = 0
+            table[table > n_bins - 1] = n_bins - 1
+        table = table.clip(0, 255).astype(np.uint8)
+        return table[ch]
+
+    channels = [tune_channel(ch) for ch in cv2.split(img)]
+    out = cv2.merge(channels)
+    return out
+
+
+def equalize_func(img):
+    '''
+        same output as PIL.ImageOps.equalize
+        PIL's implementation is different from cv2.equalize
+    '''
+    n_bins = 256
+
+    def tune_channel(ch):
+        hist = cv2.calcHist([ch], [0], None, [n_bins], [0, n_bins])
+        non_zero_hist = hist[hist != 0].reshape(-1)
+        step = np.sum(non_zero_hist[:-1]) // (n_bins - 1)
+        if step == 0:
+            return ch
+        n = np.empty_like(hist)
+        n[0] = step // 2
+        n[1:] = hist[:-1]
+        table = (np.cumsum(n) // step).clip(0, 255).astype(np.uint8)
+        return table[ch]
+
+    channels = [tune_channel(ch) for ch in cv2.split(img)]
+    out = cv2.merge(channels)
+    return out
+
+
+def rotate_func(img, degree, fill=(0, 0, 0)):
+    '''
+    like PIL, rotate by degree, not radians
+    '''
+    H, W = img.shape[0], img.shape[1]
+    center = W / 2, H / 2
+    M = cv2.getRotationMatrix2D(center, degree, 1)
+    out = cv2.warpAffine(img, M, (W, H), borderValue=fill)
+    return out
+
+
+def solarize_func(img, thresh=128):
+    '''
+        same output as PIL.ImageOps.posterize
+    '''
+    table = np.array([el if el < thresh else 255 - el for el in range(256)])
+    table = table.clip(0, 255).astype(np.uint8)
+    out = table[img]
+    return out
+
+
+def color_func(img, factor):
+    '''
+        same output as PIL.ImageEnhance.Color
+    '''
+    # implementation according to PIL definition, quite slow
+    #  degenerate = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)[:, :, np.newaxis]
+    #  out = blend(degenerate, img, factor)
+    #  M = (
+    #      np.eye(3) * factor
+    #      + np.float32([0.114, 0.587, 0.299]).reshape(3, 1) * (1. - factor)
+    #  )[np.newaxis, np.newaxis, :]
+    M = (
+        np.float32([
+            [0.886, -0.114, -0.114],
+            [-0.587, 0.413, -0.587],
+            [-0.299, -0.299, 0.701]]) * factor
+        + np.float32([[0.114], [0.587], [0.299]])
+    )
+    out = np.matmul(img, M).clip(0, 255).astype(np.uint8)
+    return out
+
+
+def contrast_func(img, factor):
+    """
+        same output as PIL.ImageEnhance.Contrast
+    """
+    mean = np.sum(np.mean(img, axis=(0, 1)) * np.array([0.114, 0.587, 0.299]))
+    table = np.array([(
+        el - mean) * factor + mean
+        for el in range(256)
+    ]).clip(0, 255).astype(np.uint8)
+    out = table[img]
+    return out
+
+
+def brightness_func(img, factor):
+    '''
+        same output as PIL.ImageEnhance.Contrast
+    '''
+    table = (np.arange(256, dtype=np.float32) *
+             factor).clip(0, 255).astype(np.uint8)
+    out = table[img]
+    return out
+
+
+def sharpness_func(img, factor):
+    '''
+    The differences the this result and PIL are all on the 4 boundaries, the center
+    areas are same
+    '''
+    kernel = np.ones((3, 3), dtype=np.float32)
+    kernel[1][1] = 5
+    kernel /= 13
+    degenerate = cv2.filter2D(img, -1, kernel)
+    if factor == 0.0:
+        out = degenerate
+    elif factor == 1.0:
+        out = img
+    else:
+        out = img.astype(np.float32)
+        degenerate = degenerate.astype(np.float32)[1:-1, 1:-1, :]
+        out[1:-1, 1:-1, :] = degenerate + factor * \
+            (out[1:-1, 1:-1, :] - degenerate)
+        out = out.astype(np.uint8)
+    return out
+
+
+def shear_x_func(img, factor, fill=(0, 0, 0)):
+    H, W = img.shape[0], img.shape[1]
+    M = np.float32([[1, factor, 0], [0, 1, 0]])
+    out = cv2.warpAffine(img, M, (W, H), borderValue=fill,
+                         flags=cv2.INTER_LINEAR).astype(np.uint8)
+    return out
+
+
+def translate_x_func(img, offset, fill=(0, 0, 0)):
+    '''
+        same output as PIL.Image.transform
+    '''
+    H, W = img.shape[0], img.shape[1]
+    M = np.float32([[1, 0, -offset], [0, 1, 0]])
+    out = cv2.warpAffine(img, M, (W, H), borderValue=fill,
+                         flags=cv2.INTER_LINEAR).astype(np.uint8)
+    return out
+
+
+def translate_y_func(img, offset, fill=(0, 0, 0)):
+    '''
+        same output as PIL.Image.transform
+    '''
+    H, W = img.shape[0], img.shape[1]
+    M = np.float32([[1, 0, 0], [0, 1, -offset]])
+    out = cv2.warpAffine(img, M, (W, H), borderValue=fill,
+                         flags=cv2.INTER_LINEAR).astype(np.uint8)
+    return out
+
+
+def posterize_func(img, bits):
+    '''
+        same output as PIL.ImageOps.posterize
+    '''
+    out = np.bitwise_and(img, np.uint8(255 << (8 - bits)))
+    return out
+
+
+def shear_y_func(img, factor, fill=(0, 0, 0)):
+    H, W = img.shape[0], img.shape[1]
+    M = np.float32([[1, 0, 0], [factor, 1, 0]])
+    out = cv2.warpAffine(img, M, (W, H), borderValue=fill,
+                         flags=cv2.INTER_LINEAR).astype(np.uint8)
+    return out
+
+
+def cutout_func(img, pad_size, replace=(0, 0, 0)):
+    replace = np.array(replace, dtype=np.uint8)
+    H, W = img.shape[0], img.shape[1]
+    rh, rw = np.random.random(2)
+    pad_size = pad_size // 2
+    ch, cw = int(rh * H), int(rw * W)
+    x1, x2 = max(ch - pad_size, 0), min(ch + pad_size, H)
+    y1, y2 = max(cw - pad_size, 0), min(cw + pad_size, W)
+    out = img.copy()
+    out[x1:x2, y1:y2, :] = replace
+    return out
+
+
+# level to args
+def enhance_level_to_args(MAX_LEVEL):
+    def level_to_args(level):
+        return ((level / MAX_LEVEL) * 1.8 + 0.1,)
+    return level_to_args
+
+
+def shear_level_to_args(MAX_LEVEL, replace_value):
+    def level_to_args(level):
+        level = (level / MAX_LEVEL) * 0.3
+        if np.random.random() > 0.5:
+            level = -level
+        return (level, replace_value)
+
+    return level_to_args
+
+
+def translate_level_to_args(translate_const, MAX_LEVEL, replace_value):
+    def level_to_args(level):
+        level = (level / MAX_LEVEL) * float(translate_const)
+        if np.random.random() > 0.5:
+            level = -level
+        return (level, replace_value)
+
+    return level_to_args
+
+
+def cutout_level_to_args(cutout_const, MAX_LEVEL, replace_value):
+    def level_to_args(level):
+        level = int((level / MAX_LEVEL) * cutout_const)
+        return (level, replace_value)
+
+    return level_to_args
+
+
+def solarize_level_to_args(MAX_LEVEL):
+    def level_to_args(level):
+        level = int((level / MAX_LEVEL) * 256)
+        return (level, )
+    return level_to_args
+
+
+def none_level_to_args(level):
+    return ()
+
+
+def posterize_level_to_args(MAX_LEVEL):
+    def level_to_args(level):
+        level = int((level / MAX_LEVEL) * 4)
+        return (level, )
+    return level_to_args
+
+
+def rotate_level_to_args(MAX_LEVEL, replace_value):
+    def level_to_args(level):
+        level = (level / MAX_LEVEL) * 30
+        if np.random.random() < 0.5:
+            level = -level
+        return (level, replace_value)
+
+    return level_to_args
+
+
+func_dict = {
+    'Identity': identity_func,
+    'AutoContrast': autocontrast_func,
+    'Equalize': equalize_func,
+    'Rotate': rotate_func,
+    'Solarize': solarize_func,
+    'Color': color_func,
+    'Contrast': contrast_func,
+    'Brightness': brightness_func,
+    'Sharpness': sharpness_func,
+    'ShearX': shear_x_func,
+    'TranslateX': translate_x_func,
+    'TranslateY': translate_y_func,
+    'Posterize': posterize_func,
+    'ShearY': shear_y_func,
+}
+
+translate_const = 10
+MAX_LEVEL = 10
+replace_value = (128, 128, 128)
+arg_dict = {
+    'Identity': none_level_to_args,
+    'AutoContrast': none_level_to_args,
+    'Equalize': none_level_to_args,
+    'Rotate': rotate_level_to_args(MAX_LEVEL, replace_value),
+    'Solarize': solarize_level_to_args(MAX_LEVEL),
+    'Color': enhance_level_to_args(MAX_LEVEL),
+    'Contrast': enhance_level_to_args(MAX_LEVEL),
+    'Brightness': enhance_level_to_args(MAX_LEVEL),
+    'Sharpness': enhance_level_to_args(MAX_LEVEL),
+    'ShearX': shear_level_to_args(MAX_LEVEL, replace_value),
+    'TranslateX': translate_level_to_args(
+        translate_const, MAX_LEVEL, replace_value
+    ),
+    'TranslateY': translate_level_to_args(
+        translate_const, MAX_LEVEL, replace_value
+    ),
+    'Posterize': posterize_level_to_args(MAX_LEVEL),
+    'ShearY': shear_level_to_args(MAX_LEVEL, replace_value),
+}
+
+
+class RandomAugment(object):
+
+    def __init__(self, N=2, M=10, isPIL=False, augs=[]):
+        self.N = N
+        self.M = M
+        self.isPIL = isPIL
+        if augs:
+            self.augs = augs
+        else:
+            self.augs = list(arg_dict.keys())
+
+    def get_random_ops(self):
+        sampled_ops = np.random.choice(self.augs, self.N)
+        return [(op, 0.5, self.M) for op in sampled_ops]
+
+    def __call__(self, img):
+        if self.isPIL:
+            img = np.array(img)
+        ops = self.get_random_ops()
+        for name, prob, level in ops:
+            if np.random.random() > prob:
+                continue
+            args = arg_dict[name](level)
+            img = func_dict[name](img, *args)
+        return img
+
+
+def build_transform(is_train, randaug=True, input_size=224, interpolation='bicubic', std_mode='IMAGENET_INCEPTION'):
+    if std_mode == 'IMAGENET_INCEPTION':
+        mean = IMAGENET_INCEPTION_MEAN
+        std = IMAGENET_INCEPTION_STD
+    elif std_mode == 'OPENAI_CLIP':
+        mean = OPENAI_CLIP_MEAN
+        std = OPENAI_CLIP_STD
+    else:
+        raise NotImplementedError
+
+    if is_train:
+        crop_scale = float(os.environ.get('TRAIN_CROP_SCALE', 0.9999))
+        t = [
+            RandomResizedCropAndInterpolation(
+                input_size, scale=(crop_scale, 1.0), interpolation='bicubic'),
+            # transforms.RandomHorizontalFlip(),
+        ]
+        if randaug and os.environ.get('TRAIN_DO_AUG', 'False') == 'True':
+            print(f'@@@@@ Do random aug during training', flush=True)
+            t.append(
+                RandomAugment(
+                    2, 7, isPIL=True,
+                    augs=[
+                        'Identity', 'AutoContrast', 'Equalize', 'Brightness', 'Sharpness',
+                        'ShearX', 'ShearY', 'TranslateX', 'TranslateY', 'Rotate',
+                    ]))
+        else:
+            print(f'@@@@@ Skip random aug during training', flush=True)
+        t += [
+            transforms.ToTensor(),
+            transforms.Normalize(mean=mean, std=std),
+        ]
+        t = transforms.Compose(t)
+    else:
+        t = transforms.Compose([
+            transforms.Resize((input_size, input_size),
+                              interpolation=transforms.InterpolationMode.BICUBIC),
+            transforms.ToTensor(),
+            transforms.Normalize(mean=mean, std=std)
+        ])
+
+    return t
+
+
+def img2b64(img_path):
+    img = Image.open(img_path)  # path to file
+    img_buffer = BytesIO()
+    img.save(img_buffer, format=img.format)
+    byte_data = img_buffer.getvalue()
+    base64_str = base64.b64encode(byte_data)  # bytes
+    base64_str = base64_str.decode("utf-8")  # str
+    return base64_str
+
+
+def str2b64(str):
+    return base64.b64encode(str.encode('utf-8')).decode('utf-8')
+
+
+def b642str(b64):
+    return base64.b64decode(b64).decode('utf-8')
+
+
+def is_dist_avail_and_initialized():
+    if not dist.is_available():
+        return False
+    if not dist.is_initialized():
+        return False
+    return True
+
+
+def get_world_size():
+    if not is_dist_avail_and_initialized():
+        return 1
+    return dist.get_world_size()
+
+
+def get_rank():
+    if not is_dist_avail_and_initialized():
+        return 0
+    return dist.get_rank()
+
+
+def all_gather(data):
+    """
+    Run all_gather on arbitrary picklable data (not necessarily tensors)
+    Args:
+        data: any picklable object
+    Returns:
+        list[data]: list of data gathered from each rank
+    """
+    world_size = get_world_size()
+    if world_size == 1:
+        return [data]
+
+    # serialized to a Tensor
+    buffer = pickle.dumps(data)
+    storage = torch.ByteStorage.from_buffer(buffer)
+    tensor = torch.ByteTensor(storage).to("cuda")
+
+    # obtain Tensor size of each rank
+    local_size = torch.LongTensor([tensor.numel()]).to("cuda")
+    size_list = [torch.LongTensor([0]).to("cuda") for _ in range(world_size)]
+    dist.all_gather(size_list, local_size)
+    size_list = [int(size.item()) for size in size_list]
+    max_size = max(size_list)
+
+    # receiving Tensor from all ranks
+    # we pad the tensor because torch all_gather does not support
+    # gathering tensors of different shapes
+    tensor_list = []
+    for _ in size_list:
+        tensor_list.append(torch.ByteTensor(size=(max_size,)).to("cuda"))
+    if local_size != max_size:
+        padding = torch.ByteTensor(size=(max_size - local_size,)).to("cuda")
+        tensor = torch.cat((tensor, padding), dim=0)
+    dist.all_gather(tensor_list, tensor)
+
+    data_list = []
+    for size, tensor in zip(size_list, tensor_list):
+        buffer = tensor.cpu().numpy().tobytes()[:size]
+        data_list.append(pickle.loads(buffer))
+
+    return data_list
+
+
+def mean(lst):
+    return sum(lst) / len(lst)
+
+
+def stop_gradient_by_name(name: str):
+    def apply_fn(module):
+        if hasattr(module, name):
+            getattr(module, name).requires_grad_(False)
+
+    return apply_fn
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/train/train_utils.py b/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/train/train_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..48af610d369071f7ec28784ed436d0f4a4e7b1c0
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/train/train_utils.py
@@ -0,0 +1,153 @@
+import os
+import gc
+import copy
+import time
+
+import torch
+import warnings
+import transformers
+
+import numpy as np
+
+from typing import Dict, Optional, Sequence
+from omnilmm import conversation as conversation_lib
+
+IGNORE_INDEX = -100
+DEFAULT_IMAGE_TOKEN = "<image>"
+DEFAULT_IMAGE_PATCH_TOKEN = "<im_patch>"
+DEFAULT_IM_START_TOKEN = "<im_start>"
+DEFAULT_IM_END_TOKEN = "<im_end>"
+
+
+def _tokenize_fn(strings: Sequence[str],
+                 tokenizer: transformers.PreTrainedTokenizer) -> Dict:
+    """Tokenize a list of strings."""
+    tokenized_list = [
+        tokenizer(
+            text,
+            return_tensors="pt",
+            padding="longest",
+            max_length=tokenizer.model_max_length,
+            truncation=True,
+        ) for text in strings
+    ]
+    input_ids = labels = [
+        tokenized.input_ids[0] for tokenized in tokenized_list
+    ]
+    input_ids_lens = labels_lens = [
+        tokenized.input_ids.ne(tokenizer.pad_token_id).sum().item()
+        for tokenized in tokenized_list
+    ]
+    return dict(
+        input_ids=input_ids,
+        labels=labels,
+        input_ids_lens=input_ids_lens,
+        labels_lens=labels_lens,
+    )
+
+
+
+def omni_preprocess(sources,
+                      tokenizer: transformers.PreTrainedTokenizer,
+                      generation=False):
+    system_content = 'You are an artificial intelligence assistant, which gives helpful, detailed, and polite answers to the human\'s questions.'
+    ignore_index = -100
+
+    response_template = '\n<|assistant|>\n'
+    instruction_template = '\n<|user|>\n'
+    response_token_ids = tokenizer.encode(
+        response_template, add_special_tokens=False)
+    instruction_token_ids = tokenizer.encode(
+        instruction_template, add_special_tokens=False)
+
+    batch_input_ids = []
+    batch_labels = []
+    for i in range(len(sources)):
+        new_source = []
+        prev_role = 'unexpect'
+        for conv_turn in sources[i]:
+            role = conv_turn['from'] if 'from' in conv_turn else conv_turn['role']
+            content = conv_turn['value'] if 'value' in conv_turn else conv_turn['content']
+
+            role = 'user' if role == 'human' else role
+            role = 'assistant' if role == 'gpt' else role
+
+            assert role in ['user', 'assistant']
+            assert role != prev_role, f'role={role}, prev_role={prev_role}'
+            prev_role = role
+
+            new_turn = {
+                'role': role,
+                'content': content
+            }
+            new_source.append(new_turn)
+        if new_source[0]['role'] != 'system':
+            new_source.insert(0, {'role': 'system', 'content': system_content})
+
+        # TODO: this automatically add '\n' to the end
+        res_text = tokenizer.apply_chat_template(
+            new_source, tokenize=False, add_generation_prompt=generation)
+        if not generation:
+            res_text = res_text.strip()
+
+        conversations_tokenized = _tokenize_fn([res_text], tokenizer)
+        res_input_ids = conversations_tokenized["input_ids"][0]
+
+        # since labels and input_ids are reference towards the same object
+        res_labels = copy.deepcopy(conversations_tokenized["labels"][0])
+
+        response_token_ids_idxs = []
+        human_token_ids_idxs = []
+
+        for assistant_idx in np.where(res_labels == response_token_ids[0])[0]:
+            # find the indexes of the start of a response.
+            if (response_token_ids == res_labels[assistant_idx: assistant_idx + len(
+                        response_token_ids)].tolist()
+                    ):
+                response_token_ids_idxs.append(
+                    assistant_idx + len(response_token_ids))
+
+        if len(response_token_ids_idxs) == 0:
+            warnings.warn(
+                f"Could not find response key `{response_template}` in the "
+                f'following instance: @===>{tokenizer.decode(res_input_ids)}<===@ '
+                f'Raw text is @===>{res_text}<===@'
+                f'Raw source is @===>{new_source}<===@'
+                f"This instance will be ignored in loss calculation. "
+                f"Note, if this happens often, consider increasing the `max_seq_length`."
+            )
+            res_labels[:] = ignore_index
+
+        human_token_ids = instruction_token_ids
+        for human_idx in np.where(res_labels == human_token_ids[0])[0]:
+            # find the indexes of the start of a human answer.
+            if human_token_ids == res_labels[human_idx: human_idx + len(human_token_ids)].tolist():
+                human_token_ids_idxs.append(human_idx)
+
+        if len(human_token_ids_idxs) == 0:
+            warnings.warn(
+                f"Could not find instruction key `{instruction_template}` in the "
+                f'following instance: @===>{tokenizer.decode(res_input_ids)}<===@ '
+                f'Raw text is @===>{res_text}<===@'
+                f'Raw source is @===>{new_source}<===@'
+                f"This instance will be ignored in loss calculation. "
+                f"Note, if this happens often, consider increasing the `max_seq_length`."
+            )
+            res_labels[:] = ignore_index
+
+        for idx, (start, end) in enumerate(zip(human_token_ids_idxs, response_token_ids_idxs)):
+            # Make pytorch loss function ignore all non response tokens
+            if idx != 0:
+                res_labels[start:end] = ignore_index
+            else:
+                res_labels[:end] = ignore_index
+
+        if len(response_token_ids_idxs) < len(human_token_ids_idxs):
+            res_labels[human_token_ids_idxs[-1]:] = ignore_index
+
+        batch_input_ids.append(res_input_ids)
+        batch_labels.append(res_labels)
+
+    return dict(input_ids=batch_input_ids, labels=batch_labels)
+
+
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/utils.py b/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..784dabe80320119b6998d79e2840303bdee3ed04
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/omnilmm/utils.py
@@ -0,0 +1,127 @@
+import datetime
+import logging
+import logging.handlers
+import os
+import sys
+
+import requests
+
+from omnilmm.constants import LOGDIR
+
+server_error_msg = "**NETWORK ERROR DUE TO HIGH TRAFFIC. PLEASE REGENERATE OR REFRESH THIS PAGE.**"
+moderation_msg = "YOUR INPUT VIOLATES OUR CONTENT MODERATION GUIDELINES. PLEASE TRY AGAIN."
+
+handler = None
+
+
+def build_logger(logger_name, logger_filename):
+    global handler
+
+    formatter = logging.Formatter(
+        fmt="%(asctime)s | %(levelname)s | %(name)s | %(message)s",
+        datefmt="%Y-%m-%d %H:%M:%S",
+    )
+
+    # Set the format of root handlers
+    if not logging.getLogger().handlers:
+        logging.basicConfig(level=logging.INFO)
+    logging.getLogger().handlers[0].setFormatter(formatter)
+
+    # Redirect stdout and stderr to loggers
+    stdout_logger = logging.getLogger("stdout")
+    stdout_logger.setLevel(logging.INFO)
+    sl = StreamToLogger(stdout_logger, logging.INFO)
+    sys.stdout = sl
+
+    stderr_logger = logging.getLogger("stderr")
+    stderr_logger.setLevel(logging.ERROR)
+    sl = StreamToLogger(stderr_logger, logging.ERROR)
+    sys.stderr = sl
+
+    # Get logger
+    logger = logging.getLogger(logger_name)
+    logger.setLevel(logging.INFO)
+
+    # Add a file handler for all loggers
+    if handler is None:
+        os.makedirs(LOGDIR, exist_ok=True)
+        filename = os.path.join(LOGDIR, logger_filename)
+        handler = logging.handlers.TimedRotatingFileHandler(
+            filename, when='D', utc=True)
+        handler.setFormatter(formatter)
+
+        for name, item in logging.root.manager.loggerDict.items():
+            if isinstance(item, logging.Logger):
+                item.addHandler(handler)
+
+    return logger
+
+
+class StreamToLogger(object):
+    """
+    Fake file-like stream object that redirects writes to a logger instance.
+    """
+
+    def __init__(self, logger, log_level=logging.INFO):
+        self.terminal = sys.stdout
+        self.logger = logger
+        self.log_level = log_level
+        self.linebuf = ''
+
+    def __getattr__(self, attr):
+        return getattr(self.terminal, attr)
+
+    def write(self, buf):
+        temp_linebuf = self.linebuf + buf
+        self.linebuf = ''
+        for line in temp_linebuf.splitlines(True):
+            # From the io.TextIOWrapper docs:
+            #   On output, if newline is None, any '\n' characters written
+            #   are translated to the system default line separator.
+            # By default sys.stdout.write() expects '\n' newlines and then
+            # translates them so this is still cross platform.
+            if line[-1] == '\n':
+                self.logger.log(self.log_level, line.rstrip())
+            else:
+                self.linebuf += line
+
+    def flush(self):
+        if self.linebuf != '':
+            self.logger.log(self.log_level, self.linebuf.rstrip())
+        self.linebuf = ''
+
+
+def disable_torch_init():
+    """
+    Disable the redundant torch default initialization to accelerate model creation.
+    """
+    import torch
+    setattr(torch.nn.Linear, "reset_parameters", lambda self: None)
+    setattr(torch.nn.LayerNorm, "reset_parameters", lambda self: None)
+
+
+def violates_moderation(text):
+    """
+    Check whether the text violates OpenAI moderation API.
+    """
+    url = "https://api.openai.com/v1/moderations"
+    headers = {"Content-Type": "application/json",
+               "Authorization": "Bearer " + os.environ["OPENAI_API_KEY"]}
+    text = text.replace("\n", "")
+    data = "{" + '"input": ' + f'"{text}"' + "}"
+    data = data.encode("utf-8")
+    try:
+        ret = requests.post(url, headers=headers, data=data, timeout=5)
+        flagged = ret.json()["results"][0]["flagged"]
+    except requests.exceptions.RequestException as e:
+        flagged = False
+    except KeyError as e:
+        flagged = False
+
+    return flagged
+
+
+def pretty_print_semaphore(semaphore):
+    if semaphore is None:
+        return "None"
+    return f"Semaphore(value={semaphore._value}, locked={semaphore.locked()})"
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/public_address_statement.md b/PyTorch/built-in/mlm/MiniCPM-V/public_address_statement.md
new file mode 100644
index 0000000000000000000000000000000000000000..af28b09849d51d93b5a0255ccbfb54d54be0f8e2
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/public_address_statement.md
@@ -0,0 +1,14 @@
+| 类型         | 开源代码地址                                                 | 文件名                                                       | 公网IP地址/公网URL地址/域名/邮箱地址                         | 用途说明     |
+| ------------ | ------------------------------------------------------------ | ------------------------------------------------------------ | ------------------------------------------------------------ | ------------ |
+| 开源代码引入 | https://github.com/OpenBMB/MiniCPM-V/blob/main/eval_mm/vlmevalkit/vlmeval/evaluate/vqa_eval.py | MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/vqa_eval.py    | https://github.com/GT-Vision-Lab/VQA                         | 模型相关说明 |
+| 开源代码引入 | https://github.com/OpenBMB/MiniCPM-V/blob/main/eval_mm/vlmevalkit/vlmeval/evaluate/vqa_eval.py | MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/vqa_eval.py    | https://github.com/google-research/pix2struct/blob/main/pix2struct/metrics.py#L81 | 源码实现     |
+| 开源代码引入 | https://github.com/OpenBMB/MiniCPM-V/blob/main/eval_mm/vlmevalkit/vlmeval/evaluate/vqa_eval.py | MiniCPM-V/eval_mm/vlmevalkit/vlmeval/evaluate/vqa_eval.py    | https://arxiv.org/pdf/2203.10244.pdf                         | 参考论文地址 |
+| 开源代码引入 | https://github.com/OpenBMB/MiniCPM-V/blob/main/eval_mm/vlmevalkit/vlmeval/api/gpt.py | MiniCPM-V/eval_mm/vlmevalkit/vlmeval/api/gpt.py              | https://api.openai.com/v1/chat/completions                   | 模型相关说明 |
+| 开源代码引入 | https://github.com/OpenBMB/MiniCPM-V/blob/main/eval_mm/vlmevalkit/vlmeval/api/gpt_int.py | MiniCPM-V/eval_mm/vlmevalkit/vlmeval/api/gpt_int.py          | http://ecs.sv.us.alles-apin.openxlab.org.cn/v1/openai/v2/text/chat | 模型相关说明 |
+| 开源代码引入 | https://github.com/OpenBMB/MiniCPM-V/blob/main/eval_mm/vlmevalkit/script/run_inference.sh | MiniCPM-V/eval_mm/vlmevalkit/script/run_inference.sh         | https://hf-mirror.com                                        | 模型相关说明 |
+| 开源代码引入 | https://github.com/OpenBMB/MiniCPM-V/blob/main/omnilmm/utils.py | MiniCPM-V/omnilmm/utils.py                                   | https://api.openai.com/v1/moderations                        | 模型相关说明 |
+| 开源代码引入 | https://github.com/OpenBMB/MiniCPM-V/blob/main/omnilmm/model/resampler.py | MiniCPM-V/omnilmm/model/resampler.py                         | https://github.com/facebookresearch/mae/blob/efb2a8062c206524e35e47d04501ed4f544c0ae8/util/pos_embed.py#L20 | 源码实现     |
+| 开源代码引入 | https://github.com/huggingface/transformers/blob/v4.44.0/src/transformers/models/idefics2/modeling_idefics2.py | MiniCPM-V/npu_monkey_patch/idefics2_conv_monkey_patch.py     | https://arxiv.org/abs/2307.06304                             | 参考论文地址 |
+| 开源代码引入 | https://github.com/huggingface/transformers/blob/v4.44.0/src/transformers/models/llama/modeling_llama.py | MiniCPM-V/npu_monkey_patch/llama_flash_attn_monkey_patch.py  | https://github.com/Dao-AILab/flash-attention/releases/tag/v2.1.0 | 源码实现     |
+| 开源代码引入 | https://github.com/huggingface/transformers/blob/v4.44.0/src/transformers/models/idefics2/modeling_idefics2.py | MiniCPM-V/npu_monkey_patch/idefics2_flash_attn_monkey_patch.py | https://github.com/Dao-AILab/flash-attention/releases/tag/v2.1.0 | 源码实现     |
+| 开源代码引入 | https://github.com/huggingface/transformers/blob/v4.44.0/src/transformers/models/llama/modeling_llama.py | MiniCPM-V/npu_monkey_patch/llama_rope_monkey_patch.py        | https://github.com/huggingface/transformers/pull/29285       | 模型相关说明 |
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/requirements.txt b/PyTorch/built-in/mlm/MiniCPM-V/requirements.txt
new file mode 100644
index 0000000000000000000000000000000000000000..dce85f998432987caa403e86c2f56245cdbacc47
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/requirements.txt
@@ -0,0 +1,33 @@
+packaging==23.2
+addict==2.4.0
+editdistance==0.6.2
+einops==0.7.0
+fairscale==0.4.0
+jsonlines==4.0.0
+markdown2==2.4.10
+matplotlib==3.7.4
+more_itertools==10.1.0
+nltk==3.8.1
+numpy==1.24.4
+opencv_python_headless==4.5.5.64
+openpyxl==3.1.2
+Pillow==10.1.0
+sacrebleu==2.3.2
+seaborn==0.13.0
+shortuuid==1.0.11
+spacy==3.7.2
+timm==0.9.10
+torch==2.1.2
+torchvision==0.16.2
+tqdm==4.66.1
+protobuf==4.25.0
+transformers==4.40.0
+typing_extensions==4.8.0
+uvicorn==0.24.0.post1
+#xformers==0.0.22.post7
+#flash_attn==2.3.4
+sentencepiece==0.1.99
+accelerate==0.30.1
+socksio==1.0.0
+gradio
+gradio_client
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/web_demo.py b/PyTorch/built-in/mlm/MiniCPM-V/web_demo.py
new file mode 100644
index 0000000000000000000000000000000000000000..668dcf4ddd9ac201d2bbe93cefcaffee1be123c1
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/web_demo.py
@@ -0,0 +1,264 @@
+#!/usr/bin/env python
+# encoding: utf-8
+import gradio as gr
+from PIL import Image
+import traceback
+import re
+import torch
+import argparse
+from transformers import AutoModel, AutoTokenizer
+
+# README, How to run demo on different devices
+# For Nvidia GPUs support BF16 (like A100, H100, RTX3090)
+# python web_demo.py --device cuda --dtype bf16
+
+# For Nvidia GPUs do NOT support BF16 (like V100, T4, RTX2080)
+# python web_demo.py --device cuda --dtype fp16
+
+# For Mac with MPS (Apple silicon or AMD GPUs).
+# PYTORCH_ENABLE_MPS_FALLBACK=1 python web_demo.py --device mps --dtype fp16
+
+# Argparser
+parser = argparse.ArgumentParser(description='demo')
+parser.add_argument('--device', type=str, default='cuda', help='cuda or mps')
+parser.add_argument('--dtype', type=str, default='bf16', help='bf16 or fp16')
+args = parser.parse_args()
+device = args.device
+assert device in ['cuda', 'mps']
+if args.dtype == 'bf16':
+    if device == 'mps':
+        print('Warning: MPS does not support bf16, will use fp16 instead')
+        dtype = torch.float16
+    else:
+        dtype = torch.bfloat16
+else:
+    dtype = torch.float16
+
+# Load model
+model_path = 'openbmb/MiniCPM-V-2'
+model = AutoModel.from_pretrained(model_path, trust_remote_code=True).to(dtype=torch.bfloat16)
+tokenizer = AutoTokenizer.from_pretrained(model_path, trust_remote_code=True)
+
+model = model.to(device=device, dtype=dtype)
+model.eval()
+
+
+
+ERROR_MSG = "Error, please retry"
+model_name = 'MiniCPM-V 2.0'
+
+form_radio = {
+    'choices': ['Beam Search', 'Sampling'],
+    #'value': 'Beam Search',
+    'value': 'Sampling',
+    'interactive': True,
+    'label': 'Decode Type'
+}
+# Beam Form
+num_beams_slider = {
+    'minimum': 0,
+    'maximum': 5,
+    'value': 3,
+    'step': 1,
+    'interactive': True,
+    'label': 'Num Beams'
+}
+repetition_penalty_slider = {
+    'minimum': 0,
+    'maximum': 3,
+    'value': 1.2,
+    'step': 0.01,
+    'interactive': True,
+    'label': 'Repetition Penalty'
+}
+repetition_penalty_slider2 = {
+    'minimum': 0,
+    'maximum': 3,
+    'value': 1.05,
+    'step': 0.01,
+    'interactive': True,
+    'label': 'Repetition Penalty'
+}
+max_new_tokens_slider = {
+    'minimum': 1,
+    'maximum': 4096,
+    'value': 1024,
+    'step': 1,
+    'interactive': True,
+    'label': 'Max New Tokens'    
+}
+
+top_p_slider = {
+    'minimum': 0,
+    'maximum': 1,
+    'value': 0.8,
+    'step': 0.05,
+    'interactive': True,
+    'label': 'Top P'    
+}
+top_k_slider = {
+    'minimum': 0,
+    'maximum': 200,
+    'value': 100,
+    'step': 1,
+    'interactive': True,
+    'label': 'Top K'    
+}
+temperature_slider = {
+    'minimum': 0,
+    'maximum': 2,
+    'value': 0.7,
+    'step': 0.05,
+    'interactive': True,
+    'label': 'Temperature'    
+}
+
+
+def create_component(params, comp='Slider'):
+    if comp == 'Slider':
+        return gr.Slider(
+            minimum=params['minimum'],
+            maximum=params['maximum'],
+            value=params['value'],
+            step=params['step'],
+            interactive=params['interactive'],
+            label=params['label']
+        )
+    elif comp == 'Radio':
+        return gr.Radio(
+            choices=params['choices'],
+            value=params['value'],
+            interactive=params['interactive'],
+            label=params['label']
+        )
+    elif comp == 'Button':
+        return gr.Button(
+            value=params['value'],
+            interactive=True
+        )
+
+
+def chat(img, msgs, ctx, params=None, vision_hidden_states=None):
+    default_params = {"num_beams":3, "repetition_penalty": 1.2, "max_new_tokens": 1024}
+    if params is None:
+        params = default_params
+    if img is None:
+        return -1, "Error, invalid image, please upload a new image", None, None
+    try:
+        image = img.convert('RGB')
+        answer, context, _ = model.chat(
+            image=image,
+            msgs=msgs,
+            context=None,
+            tokenizer=tokenizer,
+            **params
+        )
+        res = re.sub(r'(<box>.*</box>)', '', answer)
+        res = res.replace('<ref>', '')
+        res = res.replace('</ref>', '')
+        res = res.replace('<box>', '')
+        answer = res.replace('</box>', '')
+        return 0, answer, None, None
+    except Exception as err:
+        print(err)
+        traceback.print_exc()
+        return -1, ERROR_MSG, None, None
+
+
+def upload_img(image, _chatbot, _app_session):
+    image = Image.fromarray(image)
+
+    _app_session['sts']=None
+    _app_session['ctx']=[]
+    _app_session['img']=image 
+    _chatbot.append(('', 'Image uploaded successfully, you can talk to me now'))
+    return _chatbot, _app_session
+
+
+def respond(_question, _chat_bot, _app_cfg, params_form, num_beams, repetition_penalty, repetition_penalty_2, top_p, top_k, temperature):
+    if _app_cfg.get('ctx', None) is None:
+        _chat_bot.append((_question, 'Please upload an image to start'))
+        return '', _chat_bot, _app_cfg
+
+    _context = _app_cfg['ctx'].copy()
+    if _context:
+        _context.append({"role": "user", "content": _question})
+    else:
+        _context = [{"role": "user", "content": _question}] 
+    print('<User>:', _question)
+
+    if params_form == 'Beam Search':
+        params = {
+            'sampling': False,
+            'num_beams': num_beams,
+            'repetition_penalty': repetition_penalty,
+            "max_new_tokens": 896 
+        }
+    else:
+        params = {
+            'sampling': True,
+            'top_p': top_p,
+            'top_k': top_k,
+            'temperature': temperature,
+            'repetition_penalty': repetition_penalty_2,
+            "max_new_tokens": 896 
+        }
+    code, _answer, _, sts = chat(_app_cfg['img'], _context, None, params)
+    print('<Assistant>:', _answer)
+
+    _context.append({"role": "assistant", "content": _answer}) 
+    _chat_bot.append((_question, _answer))
+    if code == 0:
+        _app_cfg['ctx']=_context
+        _app_cfg['sts']=sts
+    return '', _chat_bot, _app_cfg
+
+
+def regenerate_button_clicked(_question, _chat_bot, _app_cfg, params_form, num_beams, repetition_penalty, repetition_penalty_2, top_p, top_k, temperature):
+    if len(_chat_bot) <= 1:
+        _chat_bot.append(('Regenerate', 'No question for regeneration.'))
+        return '', _chat_bot, _app_cfg
+    elif _chat_bot[-1][0] == 'Regenerate':
+        return '', _chat_bot, _app_cfg
+    else:
+        _question = _chat_bot[-1][0]
+        _chat_bot = _chat_bot[:-1]
+        _app_cfg['ctx'] = _app_cfg['ctx'][:-2]
+    return respond(_question, _chat_bot, _app_cfg, params_form, num_beams, repetition_penalty, repetition_penalty_2, top_p, top_k, temperature)
+
+
+
+with gr.Blocks() as demo:
+    with gr.Row():
+        with gr.Column(scale=1, min_width=300):
+            params_form = create_component(form_radio, comp='Radio')
+            with gr.Accordion("Beam Search") as beams_according:
+                num_beams = create_component(num_beams_slider)
+                repetition_penalty = create_component(repetition_penalty_slider)
+            with gr.Accordion("Sampling") as sampling_according:
+                top_p = create_component(top_p_slider)
+                top_k = create_component(top_k_slider)
+                temperature = create_component(temperature_slider)
+                repetition_penalty_2 = create_component(repetition_penalty_slider2)
+            regenerate = create_component({'value': 'Regenerate'}, comp='Button')
+        with gr.Column(scale=3, min_width=500):
+            app_session = gr.State({'sts':None,'ctx':None,'img':None})
+            bt_pic = gr.Image(label="Upload an image to start")
+            chat_bot = gr.Chatbot(label=f"Chat with {model_name}")
+            txt_message = gr.Textbox(label="Input text")
+            
+            regenerate.click(
+                regenerate_button_clicked,
+                [txt_message, chat_bot, app_session, params_form, num_beams, repetition_penalty, repetition_penalty_2, top_p, top_k, temperature],
+                [txt_message, chat_bot, app_session]
+            )
+            txt_message.submit(
+                respond, 
+                [txt_message, chat_bot, app_session, params_form, num_beams, repetition_penalty, repetition_penalty_2, top_p, top_k, temperature], 
+                [txt_message, chat_bot, app_session]
+            )
+            bt_pic.upload(lambda: None, None, chat_bot, queue=False).then(upload_img, inputs=[bt_pic,chat_bot,app_session], outputs=[chat_bot,app_session])
+
+# launch
+demo.launch(share=False, debug=True, show_api=False, server_port=8080, server_name="0.0.0.0")
+
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/web_demo_2.5.py b/PyTorch/built-in/mlm/MiniCPM-V/web_demo_2.5.py
new file mode 100644
index 0000000000000000000000000000000000000000..6f6b81af37f1063a2f34d13484211713728f80b6
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/web_demo_2.5.py
@@ -0,0 +1,256 @@
+#!/usr/bin/env python
+# encoding: utf-8
+import gradio as gr
+from PIL import Image
+import traceback
+import re
+import torch
+import argparse
+from transformers import AutoModel, AutoTokenizer
+
+# README, How to run demo on different devices
+
+# For Nvidia GPUs.
+# python web_demo_2.5.py --device cuda
+
+# For Mac with MPS (Apple silicon or AMD GPUs).
+# PYTORCH_ENABLE_MPS_FALLBACK=1 python web_demo_2.5.py --device mps
+
+# Argparser
+parser = argparse.ArgumentParser(description='demo')
+parser.add_argument('--device', type=str, default='cuda', help='cuda or mps')
+args = parser.parse_args()
+device = args.device
+assert device in ['cuda', 'mps']
+
+# Load model
+model_path = 'openbmb/MiniCPM-Llama3-V-2_5'
+if 'int4' in model_path:
+    if device == 'mps':
+        print('Error: running int4 model with bitsandbytes on Mac is not supported right now.')
+        exit()
+    model = AutoModel.from_pretrained(model_path, trust_remote_code=True)
+else:
+    model = AutoModel.from_pretrained(model_path, trust_remote_code=True, torch_dtype=torch.float16, device_map=device)
+tokenizer = AutoTokenizer.from_pretrained(model_path, trust_remote_code=True)
+model.eval()
+
+
+
+ERROR_MSG = "Error, please retry"
+model_name = 'MiniCPM-V 2.5'
+
+form_radio = {
+    'choices': ['Beam Search', 'Sampling'],
+    #'value': 'Beam Search',
+    'value': 'Sampling',
+    'interactive': True,
+    'label': 'Decode Type'
+}
+# Beam Form
+num_beams_slider = {
+    'minimum': 0,
+    'maximum': 5,
+    'value': 3,
+    'step': 1,
+    'interactive': True,
+    'label': 'Num Beams'
+}
+repetition_penalty_slider = {
+    'minimum': 0,
+    'maximum': 3,
+    'value': 1.2,
+    'step': 0.01,
+    'interactive': True,
+    'label': 'Repetition Penalty'
+}
+repetition_penalty_slider2 = {
+    'minimum': 0,
+    'maximum': 3,
+    'value': 1.05,
+    'step': 0.01,
+    'interactive': True,
+    'label': 'Repetition Penalty'
+}
+max_new_tokens_slider = {
+    'minimum': 1,
+    'maximum': 4096,
+    'value': 1024,
+    'step': 1,
+    'interactive': True,
+    'label': 'Max New Tokens'    
+}
+
+top_p_slider = {
+    'minimum': 0,
+    'maximum': 1,
+    'value': 0.8,
+    'step': 0.05,
+    'interactive': True,
+    'label': 'Top P'    
+}
+top_k_slider = {
+    'minimum': 0,
+    'maximum': 200,
+    'value': 100,
+    'step': 1,
+    'interactive': True,
+    'label': 'Top K'    
+}
+temperature_slider = {
+    'minimum': 0,
+    'maximum': 2,
+    'value': 0.7,
+    'step': 0.05,
+    'interactive': True,
+    'label': 'Temperature'    
+}
+
+
+def create_component(params, comp='Slider'):
+    if comp == 'Slider':
+        return gr.Slider(
+            minimum=params['minimum'],
+            maximum=params['maximum'],
+            value=params['value'],
+            step=params['step'],
+            interactive=params['interactive'],
+            label=params['label']
+        )
+    elif comp == 'Radio':
+        return gr.Radio(
+            choices=params['choices'],
+            value=params['value'],
+            interactive=params['interactive'],
+            label=params['label']
+        )
+    elif comp == 'Button':
+        return gr.Button(
+            value=params['value'],
+            interactive=True
+        )
+
+
+def chat(img, msgs, ctx, params=None, vision_hidden_states=None):
+    default_params = {"num_beams":3, "repetition_penalty": 1.2, "max_new_tokens": 1024}
+    if params is None:
+        params = default_params
+    if img is None:
+        return -1, "Error, invalid image, please upload a new image", None, None
+    try:
+        image = img.convert('RGB')
+        answer = model.chat(
+            image=image,
+            msgs=msgs,
+            tokenizer=tokenizer,
+            **params
+        )
+        res = re.sub(r'(<box>.*</box>)', '', answer)
+        res = res.replace('<ref>', '')
+        res = res.replace('</ref>', '')
+        res = res.replace('<box>', '')
+        answer = res.replace('</box>', '')
+        return 0, answer, None, None
+    except Exception as err:
+        print(err)
+        traceback.print_exc()
+        return -1, ERROR_MSG, None, None
+
+
+def upload_img(image, _chatbot, _app_session):
+    image = Image.fromarray(image)
+
+    _app_session['sts']=None
+    _app_session['ctx']=[]
+    _app_session['img']=image 
+    _chatbot.append(('', 'Image uploaded successfully, you can talk to me now'))
+    return _chatbot, _app_session
+
+
+def respond(_question, _chat_bot, _app_cfg, params_form, num_beams, repetition_penalty, repetition_penalty_2, top_p, top_k, temperature):
+    if _app_cfg.get('ctx', None) is None:
+        _chat_bot.append((_question, 'Please upload an image to start'))
+        return '', _chat_bot, _app_cfg
+
+    _context = _app_cfg['ctx'].copy()
+    if _context:
+        _context.append({"role": "user", "content": _question})
+    else:
+        _context = [{"role": "user", "content": _question}] 
+    print('<User>:', _question)
+
+    if params_form == 'Beam Search':
+        params = {
+            'sampling': False,
+            'num_beams': num_beams,
+            'repetition_penalty': repetition_penalty,
+            "max_new_tokens": 896 
+        }
+    else:
+        params = {
+            'sampling': True,
+            'top_p': top_p,
+            'top_k': top_k,
+            'temperature': temperature,
+            'repetition_penalty': repetition_penalty_2,
+            "max_new_tokens": 896 
+        }
+    code, _answer, _, sts = chat(_app_cfg['img'], _context, None, params)
+    print('<Assistant>:', _answer)
+
+    _context.append({"role": "assistant", "content": _answer}) 
+    _chat_bot.append((_question, _answer))
+    if code == 0:
+        _app_cfg['ctx']=_context
+        _app_cfg['sts']=sts
+    return '', _chat_bot, _app_cfg
+
+
+def regenerate_button_clicked(_question, _chat_bot, _app_cfg, params_form, num_beams, repetition_penalty, repetition_penalty_2, top_p, top_k, temperature):
+    if len(_chat_bot) <= 1:
+        _chat_bot.append(('Regenerate', 'No question for regeneration.'))
+        return '', _chat_bot, _app_cfg
+    elif _chat_bot[-1][0] == 'Regenerate':
+        return '', _chat_bot, _app_cfg
+    else:
+        _question = _chat_bot[-1][0]
+        _chat_bot = _chat_bot[:-1]
+        _app_cfg['ctx'] = _app_cfg['ctx'][:-2]
+    return respond(_question, _chat_bot, _app_cfg, params_form, num_beams, repetition_penalty, repetition_penalty_2, top_p, top_k, temperature)
+
+
+
+with gr.Blocks() as demo:
+    with gr.Row():
+        with gr.Column(scale=1, min_width=300):
+            params_form = create_component(form_radio, comp='Radio')
+            with gr.Accordion("Beam Search") as beams_according:
+                num_beams = create_component(num_beams_slider)
+                repetition_penalty = create_component(repetition_penalty_slider)
+            with gr.Accordion("Sampling") as sampling_according:
+                top_p = create_component(top_p_slider)
+                top_k = create_component(top_k_slider)
+                temperature = create_component(temperature_slider)
+                repetition_penalty_2 = create_component(repetition_penalty_slider2)
+            regenerate = create_component({'value': 'Regenerate'}, comp='Button')
+        with gr.Column(scale=3, min_width=500):
+            app_session = gr.State({'sts':None,'ctx':None,'img':None})
+            bt_pic = gr.Image(label="Upload an image to start")
+            chat_bot = gr.Chatbot(label=f"Chat with {model_name}")
+            txt_message = gr.Textbox(label="Input text")
+            
+            regenerate.click(
+                regenerate_button_clicked,
+                [txt_message, chat_bot, app_session, params_form, num_beams, repetition_penalty, repetition_penalty_2, top_p, top_k, temperature],
+                [txt_message, chat_bot, app_session]
+            )
+            txt_message.submit(
+                respond, 
+                [txt_message, chat_bot, app_session, params_form, num_beams, repetition_penalty, repetition_penalty_2, top_p, top_k, temperature], 
+                [txt_message, chat_bot, app_session]
+            )
+            bt_pic.upload(lambda: None, None, chat_bot, queue=False).then(upload_img, inputs=[bt_pic,chat_bot,app_session], outputs=[chat_bot,app_session])
+
+# launch
+demo.launch(share=False, debug=True, show_api=False, server_port=8080, server_name="0.0.0.0")
+
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/web_demo_streamlit-2_5.py b/PyTorch/built-in/mlm/MiniCPM-V/web_demo_streamlit-2_5.py
new file mode 100644
index 0000000000000000000000000000000000000000..e3d67e1b9b87ff26f801dfc02fa39cf6ee335130
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/web_demo_streamlit-2_5.py
@@ -0,0 +1,108 @@
+import streamlit as st
+from PIL import Image
+import torch
+from transformers import AutoModel, AutoTokenizer
+
+# Model path
+model_path = "openbmb/MiniCPM-Llama3-V-2_5"
+
+# User and assistant names
+U_NAME = "User"
+A_NAME = "Assistant"
+
+# Set page configuration
+st.set_page_config(
+    page_title="MiniCPM-Llama3-V-2_5 Streamlit",
+    page_icon=":robot:",
+    layout="wide"
+)
+
+
+# Load model and tokenizer
+@st.cache_resource
+def load_model_and_tokenizer():
+    print(f"load_model_and_tokenizer from {model_path}")
+    model = AutoModel.from_pretrained(model_path, trust_remote_code=True, torch_dtype=torch.float16).to(device="cuda")
+    tokenizer = AutoTokenizer.from_pretrained(model_path, trust_remote_code=True)
+    return model, tokenizer
+
+
+# Initialize session state
+if 'model' not in st.session_state:
+    st.session_state.model, st.session_state.tokenizer = load_model_and_tokenizer()
+    st.session_state.model.eval()
+    print("model and tokenizer had loaded completed!")
+
+# Initialize session state
+if 'chat_history' not in st.session_state:
+    st.session_state.chat_history = []
+
+# Sidebar settings
+sidebar_name = st.sidebar.title("MiniCPM-Llama3-V-2_5 Streamlit")
+max_length = st.sidebar.slider("max_length", 0, 4096, 2048, step=2)
+repetition_penalty = st.sidebar.slider("repetition_penalty", 0.0, 2.0, 1.05, step=0.01)
+top_p = st.sidebar.slider("top_p", 0.0, 1.0, 0.8, step=0.01)
+top_k = st.sidebar.slider("top_k", 0, 100, 100, step=1)
+temperature = st.sidebar.slider("temperature", 0.0, 1.0, 0.7, step=0.01)
+
+# Clear chat history button
+buttonClean = st.sidebar.button("Clear chat history", key="clean")
+if buttonClean:
+    st.session_state.chat_history = []
+    st.session_state.response = ""
+    if torch.cuda.is_available():
+        torch.cuda.empty_cache()
+    st.rerun()
+
+# Display chat history
+for i, message in enumerate(st.session_state.chat_history):
+    if message["role"] == "user":
+        with st.chat_message(name="user", avatar="user"):
+            if message["image"] is not None:
+                st.image(message["image"], caption='User uploaded image', width=448, use_column_width=False)
+                continue
+            elif message["content"] is not None:
+                st.markdown(message["content"])
+    else:
+        with st.chat_message(name="model", avatar="assistant"):
+            st.markdown(message["content"])
+
+# Select mode
+selected_mode = st.sidebar.selectbox("Select mode", ["Text", "Image"])
+if selected_mode == "Image":
+    # Image mode
+    uploaded_image = st.sidebar.file_uploader("Upload image", key=1, type=["jpg", "jpeg", "png"],
+                                              accept_multiple_files=False)
+    if uploaded_image is not None:
+        st.image(uploaded_image, caption='User uploaded image', width=468, use_column_width=False)
+        # Add uploaded image to chat history
+        st.session_state.chat_history.append({"role": "user", "content": None, "image": uploaded_image})
+
+# User input box
+user_text = st.chat_input("Enter your question")
+if user_text:
+    with st.chat_message(U_NAME, avatar="user"):
+        st.session_state.chat_history.append({"role": "user", "content": user_text, "image": None})
+        st.markdown(f"{U_NAME}: {user_text}")
+
+    # Generate reply using the model
+    model = st.session_state.model
+    tokenizer = st.session_state.tokenizer
+
+    with st.chat_message(A_NAME, avatar="assistant"):
+        # If the previous message contains an image, pass the image to the model
+        if len(st.session_state.chat_history) > 1 and st.session_state.chat_history[-2]["image"] is not None:
+            uploaded_image = st.session_state.chat_history[-2]["image"]
+            imagefile = Image.open(uploaded_image).convert('RGB')
+
+        msgs = [{"role": "user", "content": user_text}]
+        res = model.chat(image=imagefile, msgs=msgs, context=None, tokenizer=tokenizer,
+                         sampling=True, top_p=top_p, top_k=top_k, repetition_penalty=repetition_penalty,
+                         temperature=temperature, stream=True)
+
+        # Collect the generated_text str
+        generated_text = st.write_stream(res)
+
+        st.session_state.chat_history.append({"role": "model", "content": generated_text, "image": None})
+
+    st.divider()
diff --git a/PyTorch/built-in/mlm/MiniCPM-V/web_demo_streamlit.py b/PyTorch/built-in/mlm/MiniCPM-V/web_demo_streamlit.py
new file mode 100644
index 0000000000000000000000000000000000000000..204a49513315fcb0c96f1ee5d3e3ddfc8f373a65
--- /dev/null
+++ b/PyTorch/built-in/mlm/MiniCPM-V/web_demo_streamlit.py
@@ -0,0 +1,99 @@
+import streamlit as st
+from PIL import Image
+import torch
+from transformers import AutoModel, AutoTokenizer
+
+# Model path
+model_path = "openbmb/MiniCPM-V-2"
+
+# User and assistant names
+U_NAME = "User"
+A_NAME = "Assistant"
+
+# Set page configuration
+st.set_page_config(
+    page_title="Minicpm-V-2 Streamlit",
+    page_icon=":robot:",
+    layout="wide"
+)
+
+# Load model and tokenizer
+@st.cache_resource
+def load_model_and_tokenizer():
+    print(f"load_model_and_tokenizer from {model_path}")
+    model = AutoModel.from_pretrained(model_path, trust_remote_code=True, torch_dtype=torch.bfloat16).to(
+        device="cuda:0", dtype=torch.bfloat16)
+    tokenizer = AutoTokenizer.from_pretrained(model_path, trust_remote_code=True)
+    return model, tokenizer
+
+# Initialize session state
+if 'model' not in st.session_state:
+    st.session_state.model, st.session_state.tokenizer = load_model_and_tokenizer()
+    print("model and tokenizer had loaded completed!")
+
+# Initialize session state
+if 'chat_history' not in st.session_state:
+    st.session_state.chat_history = []
+
+# Sidebar settings
+sidebar_name = st.sidebar.title("Minicpm-V-2 Streamlit")
+max_length = st.sidebar.slider("max_length", 0, 4096, 2048, step=2)
+top_p = st.sidebar.slider("top_p", 0.0, 1.0, 0.8, step=0.01)
+temperature = st.sidebar.slider("temperature", 0.0, 1.0, 0.7, step=0.01)
+
+# Clear chat history button
+buttonClean = st.sidebar.button("Clear chat history", key="clean")
+if buttonClean:
+    st.session_state.chat_history = []
+    st.session_state.response = ""
+    if torch.cuda.is_available():
+        torch.cuda.empty_cache()
+    st.rerun()
+
+# Display chat history
+for i, message in enumerate(st.session_state.chat_history):
+    if message["role"] == "user":
+        with st.chat_message(name="user", avatar="user"):
+            if message["image"] is not None:
+                st.image(message["image"], caption='User uploaded image', width=468, use_column_width=False)
+                continue
+            elif message["content"] is not None:
+                st.markdown(message["content"])
+    else:
+        with st.chat_message(name="model", avatar="assistant"):
+            st.markdown(message["content"])
+
+# Select mode
+selected_mode = st.sidebar.selectbox("Select mode", ["Text", "Image"])
+if selected_mode == "Image":
+    # Image mode
+    uploaded_image = st.sidebar.file_uploader("Upload image", key=1, type=["jpg", "jpeg", "png"], accept_multiple_files=False)
+    if uploaded_image is not None:
+        st.image(uploaded_image, caption='User uploaded image', width=468, use_column_width=False)
+        # Add uploaded image to chat history
+        st.session_state.chat_history.append({"role": "user", "content": None, "image": uploaded_image})
+
+# User input box
+user_text = st.chat_input("Enter your question")
+if user_text:
+    with st.chat_message(U_NAME, avatar="user"):
+        st.session_state.chat_history.append({"role": "user", "content": user_text, "image": None})
+        st.markdown(f"{U_NAME}: {user_text}")
+
+    # Generate reply using the model
+    model = st.session_state.model
+    tokenizer = st.session_state.tokenizer
+
+    with st.chat_message(A_NAME, avatar="assistant"):
+        # If the previous message contains an image, pass the image to the model
+        if len(st.session_state.chat_history) > 1 and st.session_state.chat_history[-2]["image"] is not None:
+            uploaded_image = st.session_state.chat_history[-2]["image"]
+            imagefile = Image.open(uploaded_image).convert('RGB')
+
+        msgs = [{"role": "user", "content": user_text}]
+        res, context, _ = model.chat(image=imagefile, msgs=msgs, context=None, tokenizer=tokenizer,
+                                     sampling=True,top_p=top_p,temperature=temperature)
+        st.markdown(f"{A_NAME}: {res}")
+        st.session_state.chat_history.append({"role": "model", "content": res, "image": None})
+
+    st.divider()