Commit 05f08fe9 authored by jameskrw's avatar jameskrw
Browse files

untested updates

parent 6a8f03d4
Loading
Loading
Loading
Loading
+47 −0
Changes for vagen/examples/sokoban/debug_qwen0_5.sh: 47 added lines, 0 removed lines.
Original line number Diff line number Diff line
set -x

export VLLM_ATTENTION_BACKEND=XFORMERS

python -m vagen.env.sokoban.create_dataset --data_dir data/sokoban-text

python3 -m vagen.trainer.main_ppo \
    algorithm.adv_estimator=grpo \
    data.train_files=data/sokoban-text/train.parquet \
    data.val_files=data/sokoban-text/test.parquet \
    data.train_batch_size=16 \
    data.max_prompt_length=2 \
    data.max_response_length=2048 \
    data.image_key=images \
    actor_rollout_ref.model.path=Qwen/Qwen2.5-0.5B-Instruct \
    actor_rollout_ref.actor.optim.lr=1e-6 \
    actor_rollout_ref.model.use_remove_padding=True \
    actor_rollout_ref.actor.ppo_mini_batch_size=4 \
    actor_rollout_ref.actor.ppo_micro_batch_size_per_gpu=1 \
    actor_rollout_ref.actor.use_kl_loss=True \
    actor_rollout_ref.actor.kl_loss_coef=0.001 \
    actor_rollout_ref.actor.kl_loss_type=low_var_kl \
    actor_rollout_ref.model.enable_gradient_checkpointing=True \
    actor_rollout_ref.actor.fsdp_config.param_offload=False \
    actor_rollout_ref.actor.fsdp_config.optimizer_offload=False \
    actor_rollout_ref.rollout.log_prob_micro_batch_size_per_gpu=1 \
    actor_rollout_ref.rollout.tensor_model_parallel_size=4 \
    actor_rollout_ref.rollout.name=vllm \
    actor_rollout_ref.rollout.gpu_memory_utilization=0.6 \
    actor_rollout_ref.rollout.enable_chunked_prefill=False \
    actor_rollout_ref.rollout.enforce_eager=False \
    actor_rollout_ref.rollout.free_cache_engine=False \
    actor_rollout_ref.rollout.n=1 \
    actor_rollout_ref.ref.log_prob_micro_batch_size_per_gpu=1 \
    actor_rollout_ref.ref.fsdp_config.param_offload=True \
    algorithm.kl_ctrl.kl_coef=0.001 \
    trainer.critic_warmup=0 \
    trainer.logger=['console','wandb'] \
    trainer.project_name='vagen' \
    trainer.experiment_name='qwen2_5_05b_function_rm' \
    trainer.n_gpus_per_node=4 \
    trainer.nnodes=1 \
    trainer.save_freq=-1 \
    trainer.test_freq=5 \
    trainer.total_epochs=15 \
    +max_turns=2 \
    2>&1 | tee debug.log
+167 −0
Changes for vagen/mllm_agent/test_loss_mask.ipynb: 167 added lines, 0 removed lines.
Original line number Diff line number Diff line
%% Cell type:code id: tags:

``` python
from verl.utils import hf_tokenizer, hf_processor
import torch
```

%% Cell type:code id: tags:

``` python
text_template = "The quick brown fox jumps over the lazy dog.<image1asdasdqwa>, <image2>, <image3>."
import re
image_keys=re.findall(r'<image[a-zA-Z0-9]*>', text_template)
print(image_keys)
```

%% Output

    ['<image1asdasdqwa>', '<image2>', '<image3>']

%% Cell type:code id: tags:

``` python
model_name = "Qwen/Qwen2.5-VL-3B-Instruct"
processor = hf_processor(model_name)
tokenizer = hf_tokenizer(model_name)
```

%% Output

    Using a slow image processor as `use_fast` is unset and a slow processor was saved with this model. `use_fast=True` will be the default behavior in v4.48, even if the model was saved with a slow processor. This will result in minor differences in outputs. You'll still be able to use a slow processor with `use_fast=False`.

%% Cell type:code id: tags:

``` python
tokenizer.pad_token
```

%% Output

    '<|endoftext|>'

%% Cell type:code id: tags:

``` python
tokenizer.decode(tokenizer.pad_token_id)
```

%% Output

    '<|endoftext|>'

%% Cell type:code id: tags:

``` python
def loss_mask(input_ids, attention_mask):
    sptk_b = tokenizer.convert_tokens_to_ids('<|box_start|>')
    sptk_e = tokenizer.convert_tokens_to_ids('<|box_end|>')
    pad_token_id = tokenizer.pad_token_id

    print(f"DEBUG:input_ids.shape:{input_ids.shape}")
    batch_size = input_ids.shape[0]
    seq_len = input_ids.shape[1]

    # Initialize output tensors with same shape as inputs
    new_input_ids = input_ids.clone()
    new_attention_mask = attention_mask.clone()
    loss_mask = torch.zeros_like(input_ids)
    new_loss_mask = torch.zeros_like(input_ids)
    # Process each example in the batch
    for b in range(batch_size):
        # Count right padding tokens using attention mask
        right_pad_tokens = (new_input_ids[b] == pad_token_id).sum().item()

        # Assert that initial padding tokens have attention mask of 0
        assert torch.all(attention_mask[b, -right_pad_tokens:] == 0), "right padding tokens must have attention mask of 0"

        # Find special token indices
        sptk_b_indices = (input_ids[b] == sptk_b).nonzero().flatten()
        sptk_e_indices = (input_ids[b] == sptk_e).nonzero().flatten()

        # Create a mask for tokens that should compute loss
        hole_pos=[] # initialize holes position list with last padding token position
        for start_pos, end_pos in zip(sptk_b_indices, sptk_e_indices):
            loss_mask[b][start_pos+1:end_pos] = 1
            hole_pos.append(start_pos.item())
            hole_pos.append(end_pos.item())
        hole_pos.append(seq_len-right_pad_tokens)
        assert new_input_ids[b][seq_len-right_pad_tokens]==pad_token_id

        # shift right to fill the wholes
        holes_to_fill=1
        for i in range(0,len(hole_pos)-1):
            start_pos = hole_pos[i]
            end_pos = hole_pos[i+1]
            new_loss_mask[b,start_pos+1-holes_to_fill:end_pos-holes_to_fill]=loss_mask[b,start_pos+1:end_pos]
            new_input_ids[b,start_pos+1-holes_to_fill:end_pos-holes_to_fill]=input_ids[b,start_pos+1:end_pos]
            new_attention_mask[b,start_pos+1-holes_to_fill:end_pos-holes_to_fill]=attention_mask[b,start_pos+1:end_pos]
            holes_to_fill+=1

        valid_tokens = seq_len-right_pad_tokens-len(hole_pos)+1 # the number of non-special tokens and non-padding tokens
        new_loss_mask[b][valid_tokens:]=0
        new_input_ids[b][valid_tokens:]=pad_token_id
        new_attention_mask[b][valid_tokens:]=0

    return new_input_ids, new_attention_mask, new_loss_mask
```

%% Cell type:code id: tags:

``` python
import verl.utils.torch_functional as verl_F
import torch
prompt_with_chat_template=tokenizer.pad_token
input_ids,attention_mask=verl_F.tokenize_and_postprocess_data(prompt=prompt_with_chat_template,
                                        tokenizer=tokenizer,
                                        max_length=1024,
                                        pad_token_id=tokenizer.pad_token_id,
                                        left_pad=False,
                                        truncation="error",
                                        )
```

%% Cell type:code id: tags:

``` python
tokenizer("", return_tensors='pt', add_special_tokens=False)
```

%% Output

    {'input_ids': tensor([], size=(1, 0)), 'attention_mask': tensor([], size=(1, 0))}

%% Cell type:code id: tags:

``` python
input_ids
```

%% Output

    tensor([[151643, 151643, 151643,  ..., 151643, 151643, 151643]])

%% Cell type:code id: tags:

``` python
print(tokenizer.decode(input_ids[0]))
print(attention_mask)
```

%% Cell type:code id: tags:

``` python
new_input_ids, new_attention_mask, new_loss_mask=loss_mask(input_ids, attention_mask)
print(tokenizer.decode(new_input_ids[0][:30]))
print(new_attention_mask[0][:30])
print(new_loss_mask[0][:30])
```

%% Output

    DEBUG:input_ids.shape:torch.Size([1, 1024])
    This is a test prompt. This is a test prompt. This is a test prompt.<|endoftext|><|endoftext|><|endoftext|><|endoftext|><|endoftext|><|endoftext|><|endoftext|><|endoftext|><|endoftext|><|endoftext|><|endoftext|>
    tensor([1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0,
            0, 0, 0, 0, 0, 0])
    tensor([0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
            0, 0, 0, 0, 0, 0])