Skip to content

qwenvl提示词中使用video报错 #438

Description

@ctcnb

出bug的具体模型

qwen2.5vl

出bug的具体模型教程

Qwen2-vl-2B Lora 微调

教程负责人

李柯辰

Bug描述

import torch
import json
from transformers import Qwen2_5_VLForConditionalGeneration, AutoTokenizer, AutoProcessor, TrainingArguments, Trainer, DataCollatorForSeq2Seq
from datasets import Dataset
from qwen_vl_utils import process_vision_info
from peft import LoraConfig, get_peft_model

model_path = "Qwen/Qwen2.5-VL-3B-Instruct"
# 加载 Qwen2.5-VL-3B-Instruct
model = Qwen2_5_VLForConditionalGeneration.from_pretrained(
    model_path,
    torch_dtype=torch.bfloat16,
    device_map="auto",
)
# 加载 tokenizer 和 processor
tokenizer = AutoTokenizer.from_pretrained(model_path)
processor = AutoProcessor.from_pretrained(model_path)
# 允许梯度更新
model.enable_input_require_grads()

# 加载数据集
data_path = "../data/data.json"
train_ds = Dataset.from_json(data_path)

MAX_LENGTH = 1024

def process_func(example):
    """
    预处理输入数据,返回可被 DataCollator 接受的列表格式
    """
    output_content = example["label"]
    file_path = example['video']

    # 构造多模态对话
    messages = [
        {"role": "user", "content": [
            {"type": "video", "video": file_path, "max_pixels": 160*120, "fps": 1.0},
            {"type": "text", "text": "根据视频内容进行分类"},
        ]},
        {"role": "assistant", "content": [{"type": "text", "text": output_content}]},
    ]

    # 应用模板生成文本
    text = processor.apply_chat_template(messages, tokenize=False, add_generation_prompt=False)

    # 提取视觉输入
    image_inputs, video_inputs, video_kwargs = process_vision_info(messages, return_video_kwargs=True)
    encoding = processor(
        text=[text],
        images=image_inputs,
        videos=video_inputs,
        padding="max_length",
        truncation=True,
        max_length=MAX_LENGTH,
        return_tensors="pt",
        **video_kwargs,
    )
    encoding = encoding.to(model.device).to(model.dtype)

    inputs = {key: value.tolist() for key, value in encoding.items()}  # tensor -> list,为了方便拼接
    instruction = inputs

    response = tokenizer(f"{output_content}", add_special_tokens=False)

    input_ids = (
            instruction["input_ids"][0] + response["input_ids"] + [tokenizer.pad_token_id]
    )

    attention_mask = instruction["attention_mask"][0] + response["attention_mask"] + [1]
    labels = (
            [-100] * len(instruction["input_ids"][0])
            + response["input_ids"]
            + [tokenizer.pad_token_id]
    )


    input_ids = torch.tensor([input_ids])
    attention_mask = torch.tensor([attention_mask])
    labels = torch.tensor([labels])

    # 截断到最大长度
    if len(input_ids) > MAX_LENGTH:
        input_ids = input_ids[:MAX_LENGTH]
        attention_mask = attention_mask[:MAX_LENGTH]
        labels = labels[:MAX_LENGTH]

    return {
        "input_ids": input_ids,
        "attention_mask": attention_mask,
        "labels": labels,
        "pixel_values_videos": encoding['pixel_values_videos'],
        "video_grid_thw": encoding['video_grid_thw'],
        "second_per_grid_ts": encoding['second_per_grid_ts'],
    }

# 批量处理数据
train_dataset = train_ds.map(process_func, remove_columns=train_ds.column_names)

# LoRA 配置
config = LoraConfig(
    task_type="CAUSAL_LM",
    target_modules=["q_proj", "k_proj", "v_proj", "o_proj", "gate_proj", "up_proj", "down_proj"],
    inference_mode=False,
    r=64,
    lora_alpha=16,
    lora_dropout=0.05,
    bias="none",
)
peft_model = get_peft_model(model, config)

# 训练参数
args = TrainingArguments(
    output_dir="output/Qwen2.5-VL-LoRA",
    per_device_train_batch_size=4,
    gradient_accumulation_steps=4,
    logging_steps=10,
    save_steps=74,
    save_total_limit=3,
    num_train_epochs=5,
    learning_rate=1e-4,
    gradient_checkpointing=True,
    report_to=[],
)

# Trainer
trainer = Trainer(
    model=peft_model,
    args=args,
    train_dataset=train_dataset,
    data_collator=DataCollatorForSeq2Seq(tokenizer=tokenizer, padding=True),
)

# 开始训练
trainer.train()

[2025-07-09 15:06:00,982] [INFO] [logging.py:107:log_dist] [Rank -1] [TorchCheckpointEngine] Initialized with serialization = False
No label_names provided for model class PeftModelForCausalLM. Since PeftModel hides base models input arguments, if label_names is not given, label_names can't be set automatically within Trainer. Note that empty label_names list will be used instead.
0%| | 0/5 [00:00<?, ?it/s]Traceback (most recent call last):
File "/home/jizhi/anaconda3/envs/llm/lib/python3.10/site-packages/transformers/tokenization_utils_base.py", line 778, in convert_to_tensors
tensor = as_tensor(value)
File "/home/jizhi/anaconda3/envs/llm/lib/python3.10/site-packages/transformers/tokenization_utils_base.py", line 740, in as_tensor
return torch.tensor(value)
ValueError: expected sequence of length 160 at dim 1 (got 192)

The above exception was the direct cause of the following exception:

Traceback (most recent call last):
File "/home/jizhi/code/ad/train/qwen-vl-hmdb51.py", line 134, in
trainer.train()
File "/home/jizhi/anaconda3/envs/llm/lib/python3.10/site-packages/transformers/trainer.py", line 2240, in train
return inner_training_loop(
File "/home/jizhi/anaconda3/envs/llm/lib/python3.10/site-packages/transformers/trainer.py", line 2509, in _inner_training_loop
batch_samples, num_items_in_batch = self.get_batch_samples(epoch_iterator, num_batches, args.device)
File "/home/jizhi/anaconda3/envs/llm/lib/python3.10/site-packages/transformers/trainer.py", line 5263, in get_batch_samples
batch_samples.append(next(epoch_iterator))
File "/home/jizhi/anaconda3/envs/llm/lib/python3.10/site-packages/accelerate/data_loader.py", line 566, in iter
current_batch = next(dataloader_iter)
File "/home/jizhi/anaconda3/envs/llm/lib/python3.10/site-packages/torch/utils/data/dataloader.py", line 708, in next
data = self._next_data()
File "/home/jizhi/anaconda3/envs/llm/lib/python3.10/site-packages/torch/utils/data/dataloader.py", line 764, in _next_data
data = self._dataset_fetcher.fetch(index) # may raise StopIteration
File "/home/jizhi/anaconda3/envs/llm/lib/python3.10/site-packages/torch/utils/data/_utils/fetch.py", line 55, in fetch
return self.collate_fn(data)
File "/home/jizhi/anaconda3/envs/llm/lib/python3.10/site-packages/transformers/data/data_collator.py", line 683, in call
batch = pad_without_fast_tokenizer_warning(
File "/home/jizhi/anaconda3/envs/llm/lib/python3.10/site-packages/transformers/data/data_collator.py", line 67, in pad_without_fast_tokenizer_warning
padded = tokenizer.pad(*pad_args, **pad_kwargs)
File "/home/jizhi/anaconda3/envs/llm/lib/python3.10/site-packages/transformers/tokenization_utils_base.py", line 3386, in pad
return BatchEncoding(batch_outputs, tensor_type=return_tensors)
File "/home/jizhi/anaconda3/envs/llm/lib/python3.10/site-packages/transformers/tokenization_utils_base.py", line 242, in init
self.convert_to_tensors(tensor_type=tensor_type, prepend_batch_axis=prepend_batch_axis)
File "/home/jizhi/anaconda3/envs/llm/lib/python3.10/site-packages/transformers/tokenization_utils_base.py", line 794, in convert_to_tensors
raise ValueError(
ValueError: Unable to create tensor, you should probably activate truncation and/or padding with 'padding=True' 'truncation=True' to have batched tensors with the same length. Perhaps your features (pixel_values_videos in this case) have excessive nesting (inputs type list where type int is expected).
0%| | 0/5 [00:00<?, ?it/s]
ERROR conda.cli.main_run:execute(49): conda run python /home/jizhi/code/ad/train/qwen-vl-hmdb51.py failed. (See above for error)

进程已结束,退出代码为 1

复现步骤

运行代码

期望行为

正常运行

环境信息

Package Version


absl-py 2.1.0
accelerate 1.6.0
aiohappyeyeballs 2.6.1
aiohttp 3.11.13
aiosignal 1.3.2
annotated-types 0.7.0
anyio 4.8.0
async-timeout 5.0.1
attrs 25.2.0
audioread 3.0.1
av 14.4.0
binpacking 1.5.2
bitsandbytes 0.45.3
boto3 1.38.4
botocore 1.38.4
certifi 2025.1.31
cffi 1.17.1
charset-normalizer 3.4.1
click 8.1.8
colorama 0.4.6
cut-cross-entropy 25.1.1
datasets 3.5.1
decorator 5.2.1
decord 0.6.0
deepspeed 0.17.1
diffusers 0.32.2
dill 0.3.8
distro 1.9.0
docstring_parser 0.16
editdistance 0.8.1
einops 0.8.1
exceptiongroup 1.2.2
fastapi 0.115.1
ffmpeg-python 0.2.0
filelock 3.17.0
flash_attn 2.8.0.post2
frozenlist 1.5.0
fsspec 2024.2.0
future 1.0.0
gitdb 4.0.12
GitPython 3.1.44
grpcio 1.71.0
h11 0.16.0
hf_transfer 0.1.9
hjson 3.1.0
httpcore 1.0.9
httpx 0.28.1
huggingface-hub 0.30.2
idna 3.10
importlib_metadata 8.6.1
jieba 0.42.1
Jinja2 3.1.6
jiter 0.9.0
jmespath 1.0.1
joblib 1.4.2
lazy_loader 0.4
librosa 0.11.0
llvmlite 0.44.0
Markdown 3.7
markdown-it-py 3.0.0
MarkupSafe 3.0.2
mdurl 0.1.2
modelscope 1.25.0
mpmath 1.3.0
msgpack 1.1.1
msgspec 0.19.0
multidict 6.1.0
multiprocess 0.70.16
networkx 3.4.2
ninja 1.11.1.4
nltk 3.9.1
numba 0.61.2
numpy 2.1.0
nvidia-cublas-cu12 12.4.5.8
nvidia-cuda-cupti-cu12 12.4.127
nvidia-cuda-nvrtc-cu12 12.4.127
nvidia-cuda-runtime-cu12 12.4.127
nvidia-cudnn-cu12 9.1.0.70
nvidia-cufft-cu12 11.2.1.3
nvidia-curand-cu12 10.3.5.147
nvidia-cusolver-cu12 11.6.1.9
nvidia-cusparse-cu12 12.3.1.170
nvidia-cusparselt-cu12 0.6.2
nvidia-ml-py 12.570.86
nvidia-ml-py3 7.352.0
nvidia-nccl-cu12 2.21.5
nvidia-nvjitlink-cu12 12.4.127
nvidia-nvtx-cu12 12.4.127
openai 1.76.2
opencv-python 4.11.0
opencv-python-headless 4.11.0
packaging 24.2
pandas 2.2.3
peft 0.15.2
pillow 11.1.0
pip 25.0.1
platformdirs 4.3.8
pooch 1.8.2
propcache 0.3.0
protobuf 3.20.3
psutil 7.0.0
py-cpuinfo 9.0.0
pyarrow 19.0.1
pyarrow-hotfix 0.7
pycparser 2.22
pydantic 2.10.6
pydantic_core 2.27.2
Pygments 2.19.1
pynvml 12.0.0
python-dateutil 2.9.0.post0
pytz 2025.1
PyYAML 6.0.2
qwen-omni-utils 0.0.8
qwen-vl-utils 0.0.11
regex 2024.11.6
requests 2.32.3
rich 13.9.4
s3transfer 0.12.0
safetensors 0.5.3
scikit-learn 1.6.1
scipy 1.15.2
sentencepiece 0.2.0
sentry-sdk 2.32.0
setuptools 75.8.2
shtab 1.7.1
six 1.17.0
smmap 5.0.2
sniffio 1.3.1
soundfile 0.13.1
soxr 0.5.0.post1
starlette 0.38.6
swankit 0.1.7
swanlab 0.5.7
sympy 1.13.1
tensorboard 2.19.0
tensorboard-data-server 0.7.2
threadpoolctl 3.5.0
tokenizers 0.21.2
torch 2.6.0
TorchCodec 0.2.1+cu118
torchvision 0.21.0
tqdm 4.67.1
transformers 4.52.3
triton 3.2.0
trl 0.15.2
typeguard 4.4.2
typing_extensions 4.12.2
tyro 0.9.16
tzdata 2025.1
unsloth 2025.5.9
unsloth_zoo 2025.5.11
urllib3 2.3.0
uvicorn 0.30.6
wandb 0.21.0
Werkzeug 3.1.3
wheel 0.45.1
xformers 0.0.29.post3
xxhash 3.5.0
yarl 1.18.3
zipp 3.21.0
(llm) jizhi@jizhi:~$

其他信息

无

确认事项 / Verification

  • 此问题未在过往Issue中被报告过 / This issue hasn't been reported before

Activity

Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment

Metadata

Metadata

Assignees

No one assigned

    Labels

    bugSomething isn't working

    Type

    No type

    Projects

    No projects

      Milestone

      No milestone

      Relationships

      None yet

      Development

      No branches or pull requests

      Issue actions