refactor data preprocessing, fix mllm rlhf

Former-commit-id: 53ff2dd24f9121ea30c95063bb72e49a9b31e980
2024-05-24 04:08:25 +08:00
parent 1078611259
commit bf59383783
15 changed files with 572 additions and 464 deletions
--- a/src/llamafactory/chat/vllm_engine.py
+++ b/src/llamafactory/chat/vllm_engine.py
@@ -98,7 +98,7 @@ class VllmEngine(BaseEngine):
            and image is not None
            and not hasattr(self.processor, "image_seq_length")
            and IMAGE_TOKEN not in messages[0]["content"]
-        ):  # llava case
+        ):  # llava-like models
            messages[0]["content"] = IMAGE_TOKEN * self.image_feature_size + messages[0]["content"]

        paired_messages = messages + [{"role": "assistant", "content": ""}]