keentomato commited on Mar 12

Commit

95c9918

verified ·

1 Parent(s): 01ac2fa

Upload SFT checkpoint step=484

Browse files

Files changed (21) hide show

.gitattributes +1 -0
README.md +168 -0
adapters.bin +3 -0
added_tokens.json +24 -0
chat_template.jinja +7 -0
config.json +136 -0
generation_config.json +4 -0
heads.bin +3 -0
merges.txt +0 -0
model-00001-of-00004.safetensors +3 -0
model-00002-of-00004.safetensors +3 -0
model-00003-of-00004.safetensors +3 -0
model-00004-of-00004.safetensors +3 -0
model.safetensors.index.json +0 -0
preprocessor_config.json +31 -0
special_tokens_map.json +38 -0
tokenizer.json +3 -0
tokenizer_config.json +222 -0
training_meta.json +12 -0
video_preprocessor_config.json +55 -0
vocab.json +0 -0

.gitattributes CHANGED Viewed

@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text

 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
+tokenizer.json filter=lfs diff=lfs merge=lfs -text

README.md ADDED Viewed

	@@ -0,0 +1,168 @@

+---
+language: en
+license: apache-2.0
+tags:
+  - human-behavior
+  - multimodal
+  - qwen2.5-omni
+  - sarcasm-detection
+  - sarcasm
+datasets:
+  - keentomato/human_behavior_atlas
+---
+# OmniSapiens BAM — Sarcasm Detection
+Fine-tuned [Qwen2.5-Omni-7B](https://huggingface.co/Qwen/Qwen2.5-Omni-7B) for multimodal sarcasm detection on the MUStARD/MMSD benchmark. Uses LoRA adapters merged into the backbone and a lightweight classification head.
+## Benchmark
+Evaluated on [keentomato/human_behavior_atlas](https://huggingface.co/datasets/keentomato/human_behavior_atlas).
+## Usage
+### Installation
+```bash
+pip install transformers torch huggingface_hub
+```
+### Classification
+```python
+import json, torch
+from huggingface_hub import hf_hub_download
+from transformers import Qwen2_5OmniThinkerForConditionalGeneration, AutoProcessor
+MODEL_ID = "keentomato/omnisapiens_bam_sarcasm_detection"
+# 1. Load backbone and processor
+model = Qwen2_5OmniThinkerForConditionalGeneration.from_pretrained(
+    MODEL_ID, torch_dtype=torch.float16, device_map="auto"
+)
+processor = AutoProcessor.from_pretrained(MODEL_ID)
+# 2. Load classification heads and label scheme
+heads_path = hf_hub_download(MODEL_ID, "heads.bin")
+label_path = hf_hub_download(MODEL_ID, "label_scheme.json")
+heads_sd = torch.load(heads_path, map_location="cpu")
+with open(label_path) as f:
+    label_scheme = json.load(f)
+# 3. Reconstruct domain heads
+global_classes = label_scheme["meta"]["global_classes"]  # {domain: [{index, label}, ...]}
+hidden_size = model.config.hidden_size
+domain_names = list(global_classes.keys())
+domain_heads = torch.nn.ModuleList([
+    torch.nn.Linear(hidden_size, len(global_classes[d])) for d in domain_names
+])
+domain_heads.load_state_dict({k.replace("heads.", ""): v for k, v in heads_sd.items()})
+domain_heads.eval().to(model.device).to(torch.float16)
+domain_to_id = {d: i for i, d in enumerate(domain_names)}
+# 4. Prepare multimodal inputs
+# video_tensor: [T, C, H, W] tensor or list of PIL images
+# audio_waveform: 1-D numpy array / tensor at 16 kHz
+domain = "sarcasm"
+messages = [{"role": "user", "content": [
+    {"type": "video"},
+    {"type": "audio"},
+    {"type": "text", "text": "Classify the human behavior expressed."},
+]}]
+text = processor.apply_chat_template(messages, add_generation_prompt=False, tokenize=False)
+inputs = processor(text=[text], videos=[video_tensor], audio=[audio_waveform], return_tensors="pt")
+inputs = {k: v.to(model.device) for k, v in inputs.items()}
+# 5. Forward pass — pool penultimate hidden layer, route through domain head
+with torch.no_grad():
+    out = model(**inputs, output_hidden_states=True, use_cache=False)
+    h = out.hidden_states[-2]                            # [B, T, H]
+    mask = inputs["attention_mask"].unsqueeze(-1).float()
+    pooled = (h * mask).sum(1) / mask.sum(1)             # [B, H]
+    logits = domain_heads[domain_to_id[domain]](pooled.float())  # [B, K_d]
+    pred_idx = logits.argmax(dim=-1).item()
+label_name = global_classes[domain][pred_idx]["label"]
+print(f"Predicted {domain}: {label_name}")
+```
+### Behavioral Descriptors (BAM Adapters)
+As `adapters.bin` is present in the repo, the model supports side-channel
+behavioral descriptors extracted from OpenPose (video) and OpenSmile (audio).
+These replace the raw video/audio inputs to the backbone with pre-computed
+behavioral feature vectors that are injected via lightweight MLP adapters.
+**Video — OpenPose keypoints**
+OpenPose produces a dict per clip with keys `pose`, `face`, `left_hand`, `right_hand`,
+each a `[T, K, 2or3]` tensor (T frames, K keypoints, x/y/conf).
+```python
+def prepare_video_feats(openpose_dict, temporal_mode="meanstd"):
+    """OpenPose dict → pooled feature vector [D_v_pooled]."""
+    parts = []
+    for key in ("pose", "face", "left_hand", "right_hand"):
+        t = openpose_dict.get(key)  # [T, K, 2or3]
+        if t is None: continue
+        t = torch.as_tensor(t).float()[..., :2]  # drop confidence, keep x/y
+        parts.append(t.reshape(t.shape[0], -1))  # [T, K*2]
+    seq = torch.cat(parts, dim=-1).float()       # [T, D_v]
+    if temporal_mode == "meanstd":
+        return torch.cat([seq.mean(0), seq.std(0)])  # [D_v*2]
+    return seq.mean(0)                               # [D_v]
+video_feats = prepare_video_feats(openpose_dict).unsqueeze(0)  # [1, D_v_pooled]
+```
+**Audio — OpenSmile features**
+OpenSmile produces a dict with key `features` → `[T, D_a]` or `[D_a]`.
+```python
+def prepare_audio_feats(opensmile_dict):
+    """OpenSmile dict → L2-normalised feature vector [D_a]."""
+    x = torch.as_tensor(opensmile_dict["features"]).float()
+    if x.ndim == 2: x = x.squeeze(0)  # [D_a] (single frame assumed)
+    return x / x.norm(p=2).clamp_min(1e-6)
+audio_feats = prepare_audio_feats(opensmile_dict).unsqueeze(0)  # [1, D_a]
+```
+**Loading and applying the adapters**
+```python
+import torch, torch.nn as nn
+from huggingface_hub import hf_hub_download
+adapters_sd = torch.load(hf_hub_download(MODEL_ID, "adapters.bin"), map_location="cpu")
+# Infer architecture from saved weight shapes — no config needed
+def _make_adapter(prefix, sd):
+    w0 = sd[f"{prefix}.mlp.0.weight"]          # [hidden, feat_dim]
+    w2 = sd[f"{prefix}.mlp.2.weight"]          # [out_dim, hidden]
+    feat_dim, hidden, out_dim = w0.shape[1], w0.shape[0], w2.shape[0]
+    mlp = nn.Sequential(nn.Linear(feat_dim, hidden), nn.ReLU(), nn.Linear(hidden, out_dim))
+    alpha = nn.Parameter(sd[f"{prefix}.alpha"])
+    class _Adapter(nn.Module):
+        def __init__(self): super().__init__(); self.mlp = mlp; self.alpha = alpha
+        def forward(self, x): return self.mlp(x) * self.alpha
+    m = _Adapter()
+    m.load_state_dict({k[len(prefix)+1:]: v for k, v in sd.items() if k.startswith(prefix)}, strict=False)
+    return m.eval()
+video_adapter = _make_adapter("video_adapter", adapters_sd).to(model.device).half()
+audio_adapter = _make_adapter("audio_adapter", adapters_sd).to(model.device).half()
+# Augment pooled repr with BAM deltas before the classification head
+with torch.no_grad():
+    out = model(**inputs, output_hidden_states=True, use_cache=False)
+    h = out.hidden_states[-2]
+    mask = inputs["attention_mask"].unsqueeze(-1).float()
+    pooled = (h * mask).sum(1) / mask.sum(1)                    # [B, H]
+    pooled = pooled + video_adapter(video_feats.to(model.device).half())
+    pooled = pooled + audio_adapter(audio_feats.to(model.device).half())
+    logits = domain_heads[domain_to_id[domain]](pooled.float())
+    pred_idx = logits.argmax(dim=-1).item()
+label_name = global_classes[domain][pred_idx]["label"]
+print(f"Predicted {domain}: {label_name}")
+```

adapters.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:b623a6887358ddbf98f7b9599c702e5d877000183088e0fa799475a2aebbebf6
+size 27584645

added_tokens.json ADDED Viewed

	@@ -0,0 +1,24 @@

+{
+  "</tool_call>": 151658,
+  "<tool_call>": 151657,
+  "<|AUDIO|>": 151646,
+  "<|IMAGE|>": 151655,
+  "<|VIDEO|>": 151656,
+  "<|audio_bos|>": 151647,
+  "<|audio_eos|>": 151648,
+  "<|box_end|>": 151649,
+  "<|endoftext|>": 151643,
+  "<|file_sep|>": 151664,
+  "<|fim_middle|>": 151660,
+  "<|fim_pad|>": 151662,
+  "<|fim_prefix|>": 151659,
+  "<|fim_suffix|>": 151661,
+  "<|im_end|>": 151645,
+  "<|im_start|>": 151644,
+  "<|quad_end|>": 151651,
+  "<|quad_start|>": 151650,
+  "<|repo_name|>": 151663,
+  "<|vision_bos|>": 151652,
+  "<|vision_eos|>": 151653,
+  "<|vision_pad|>": 151654
+}

chat_template.jinja ADDED Viewed

	@@ -0,0 +1,7 @@

+{% set audio_count = namespace(value=0) %}{% set image_count = namespace(value=0) %}{% set video_count = namespace(value=0) %}{% for message in messages %}{% if loop.first and message['role'] != 'system' %}<|im_start|>system
+You are a helpful assistant.<|im_end|>
+{% endif %}<|im_start|>{{ message['role'] }}
+{% if message['content'] is string %}{{ message['content'] }}<|im_end|>
+{% else %}{% for content in message['content'] %}{% if content['type'] == 'image' or 'image' in content or 'image_url' in content %}{% set image_count.value = image_count.value + 1 %}{% if add_vision_id %}Picture {{ image_count.value }}: {% endif %}<|vision_bos|><|IMAGE|><|vision_eos|>{% elif content['type'] == 'audio' or 'audio' in content or 'audio_url' in content %}{% set audio_count.value = audio_count.value + 1 %}{% if add_audio_id %}Audio {{ audio_count.value }}: {% endif %}<|audio_bos|><|AUDIO|><|audio_eos|>{% elif content['type'] == 'video' or 'video' in content %}{% set video_count.value = video_count.value + 1 %}{% if add_vision_id %}Video {{ video_count.value }}: {% endif %}<|vision_bos|><|VIDEO|><|vision_eos|>{% elif 'text' in content %}{{ content['text'] }}{% endif %}{% endfor %}<|im_end|>
+{% endif %}{% endfor %}{% if add_generation_prompt %}<|im_start|>assistant
+{% endif %}

config.json ADDED Viewed

	@@ -0,0 +1,136 @@

+{
+  "_attn_implementation_autoset": true,
+  "architectures": [
+    "Qwen2_5OmniThinkerForConditionalGeneration"
+  ],
+  "audio_config": {
+    "_attn_implementation_autoset": true,
+    "activation_dropout": 0.0,
+    "activation_function": "gelu",
+    "attention_dropout": 0.0,
+    "d_model": 1280,
+    "dropout": 0.0,
+    "encoder_attention_heads": 20,
+    "encoder_ffn_dim": 5120,
+    "encoder_layerdrop": 0.0,
+    "encoder_layers": 32,
+    "init_std": 0.02,
+    "initializer_range": 0.02,
+    "max_source_positions": 1500,
+    "model_type": "qwen2_5_omni_audio_encoder",
+    "n_window": 100,
+    "num_hidden_layers": 32,
+    "num_mel_bins": 128,
+    "output_dim": 3584,
+    "scale_embedding": false,
+    "torch_dtype": "float16"
+  },
+  "audio_end_token_id": 151648,
+  "audio_start_token_id": 151647,
+  "audio_token_index": 151646,
+  "bos_token_id": 151644,
+  "eos_token_id": 151645,
+  "ignore_index": -100,
+  "image_token_index": 151655,
+  "init_std": 0.02,
+  "initializer_range": 0.02,
+  "model_type": "qwen2_5_omni_thinker",
+  "pad_token_id": 151643,
+  "position_id_per_seconds": 25,
+  "seconds_per_chunk": 2,
+  "text_config": {
+    "attention_dropout": 0.0,
+    "hidden_act": "silu",
+    "hidden_size": 3584,
+    "init_std": 0.02,
+    "initializer_range": 0.02,
+    "intermediate_size": 18944,
+    "layer_types": [
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention"
+    ],
+    "max_position_embeddings": 32768,
+    "max_window_layers": 28,
+    "model_type": "qwen2_5_omni_text",
+    "num_attention_heads": 28,
+    "num_hidden_layers": 28,
+    "num_key_value_heads": 4,
+    "rms_norm_eps": 1e-06,
+    "rope_scaling": {
+      "mrope_section": [
+        16,
+        24,
+        24
+      ],
+      "rope_type": "default",
+      "type": "default"
+    },
+    "rope_theta": 1000000.0,
+    "sliding_window": null,
+    "torch_dtype": "float16",
+    "use_cache": true,
+    "use_sliding_window": false,
+    "vocab_size": 152064
+  },
+  "torch_dtype": "float16",
+  "transformers_version": "4.55.2",
+  "user_token_id": 872,
+  "video_token_index": 151656,
+  "vision_config": {
+    "_attn_implementation_autoset": true,
+    "depth": 32,
+    "embed_dim": 1280,
+    "fullatt_block_indexes": [
+      7,
+      15,
+      23,
+      31
+    ],
+    "hidden_act": "silu",
+    "hidden_size": 1280,
+    "in_channels": 3,
+    "in_chans": 3,
+    "init_std": 0.02,
+    "initializer_range": 0.02,
+    "intermediate_size": 3420,
+    "model_type": "qwen2_5_omni_vision_encoder",
+    "num_heads": 16,
+    "out_hidden_size": 3584,
+    "patch_size": 14,
+    "spatial_merge_size": 2,
+    "spatial_patch_size": 14,
+    "temporal_patch_size": 2,
+    "tokens_per_second": 25,
+    "torch_dtype": "float16",
+    "window_size": 112
+  },
+  "vision_end_token_id": 151653,
+  "vision_start_token_id": 151652,
+  "vision_token_id": 151654
+}

generation_config.json ADDED Viewed

	@@ -0,0 +1,4 @@

+{
+  "_from_model_config": true,
+  "transformers_version": "4.55.2"
+}

heads.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:0743f3b423f74b19fbaf98daf6f163e07e72dc6af39e12ac7794336bf6aa07de
+size 363217

merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

model-00001-of-00004.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:9b23263d80740203786a9b9848018cf9a184b831799fe63dba2f9cbd42506419
+size 4985046168

model-00002-of-00004.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:aafa0787d313f3c71830aa2c02b6499847212a16b72851568bb83da69a801dba
+size 4991495656

model-00003-of-00004.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:1cc320a70463d10173a163ab6bfd8edc93a998721e9cf22884cffa50ad91d70c
+size 4991495752

model-00004-of-00004.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:b1f300d054c6b0640a01cb29af87c09adcd5a878fd47e04b4e4efde4cb9db467
+size 2895739680

model.safetensors.index.json ADDED Viewed

The diff for this file is too large to render. See raw diff

preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,31 @@

+{
+  "chunk_length": 300,
+  "dither": 0.0,
+  "feature_extractor_type": "WhisperFeatureExtractor",
+  "feature_size": 128,
+  "hop_length": 160,
+  "image_mean": [
+    0.48145466,
+    0.4578275,
+    0.40821073
+  ],
+  "image_processor_type": "Qwen2VLImageProcessor",
+  "image_std": [
+    0.26862954,
+    0.26130258,
+    0.27577711
+  ],
+  "max_pixels": 12845056,
+  "merge_size": 2,
+  "min_pixels": 3136,
+  "n_fft": 400,
+  "n_samples": 4800000,
+  "nb_max_frames": 30000,
+  "padding_side": "right",
+  "padding_value": 0.0,
+  "patch_size": 14,
+  "processor_class": "Qwen2_5OmniProcessor",
+  "return_attention_mask": true,
+  "sampling_rate": 16000,
+  "temporal_patch_size": 2
+}

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,38 @@

+{
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|AUDIO|>",
+    "<|audio_bos|>",
+    "<|audio_eos|>",
+    "<|box_end|>",
+    "<|quad_start|>",
+    "<|quad_end|>",
+    "<|vision_bos|>",
+    "<|vision_eos|>",
+    "<|vision_pad|>",
+    "<|IMAGE|>",
+    "<|VIDEO|>"
+  ],
+  "audio_bos_token": "<|audio_bos|>",
+  "audio_eos_token": "<|audio_eos|>",
+  "audio_token": "<|AUDIO|>",
+  "eos_token": {
+    "content": "<|im_end|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "image_token": "<|IMAGE|>",
+  "pad_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "video_token": "<|VIDEO|>",
+  "vision_bos_token": "<|vision_bos|>",
+  "vision_eos_token": "<|vision_eos|>"
+}

tokenizer.json ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:8441917e39ae0244e06d704b95b3124795cec478e297f9afac39ba670d7e9d99
+size 11421870

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,222 @@

+{
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "151643": {
+      "content": "<|endoftext|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151644": {
+      "content": "<|im_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151645": {
+      "content": "<|im_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151646": {
+      "content": "<|AUDIO|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151647": {
+      "content": "<|audio_bos|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151648": {
+      "content": "<|audio_eos|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151649": {
+      "content": "<|box_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151650": {
+      "content": "<|quad_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151651": {
+      "content": "<|quad_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151652": {
+      "content": "<|vision_bos|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151653": {
+      "content": "<|vision_eos|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151654": {
+      "content": "<|vision_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151655": {
+      "content": "<|IMAGE|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151656": {
+      "content": "<|VIDEO|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151657": {
+      "content": "<tool_call>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151658": {
+      "content": "</tool_call>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151659": {
+      "content": "<|fim_prefix|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151660": {
+      "content": "<|fim_middle|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151661": {
+      "content": "<|fim_suffix|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151662": {
+      "content": "<|fim_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151663": {
+      "content": "<|repo_name|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151664": {
+      "content": "<|file_sep|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    }
+  },
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|AUDIO|>",
+    "<|audio_bos|>",
+    "<|audio_eos|>",
+    "<|box_end|>",
+    "<|quad_start|>",
+    "<|quad_end|>",
+    "<|vision_bos|>",
+    "<|vision_eos|>",
+    "<|vision_pad|>",
+    "<|IMAGE|>",
+    "<|VIDEO|>"
+  ],
+  "audio_bos_token": "<|audio_bos|>",
+  "audio_eos_token": "<|audio_eos|>",
+  "audio_token": "<|AUDIO|>",
+  "bos_token": null,
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<|im_end|>",
+  "errors": "replace",
+  "extra_special_tokens": {
+    "audio_bos_token": "<|audio_bos|>",
+    "audio_eos_token": "<|audio_eos|>",
+    "audio_token": "<|AUDIO|>",
+    "image_token": "<|IMAGE|>",
+    "video_token": "<|VIDEO|>",
+    "vision_bos_token": "<|vision_bos|>",
+    "vision_eos_token": "<|vision_eos|>"
+  },
+  "image_token": "<|IMAGE|>",
+  "model_max_length": 32768,
+  "pad_token": "<|endoftext|>",
+  "processor_class": "Qwen2_5OmniProcessor",
+  "split_special_tokens": false,
+  "tokenizer_class": "Qwen2Tokenizer",
+  "unk_token": null,
+  "video_token": "<|VIDEO|>",
+  "vision_bos_token": "<|vision_bos|>",
+  "vision_eos_token": "<|vision_eos|>"
+}

training_meta.json ADDED Viewed

	@@ -0,0 +1,12 @@

+{
+  "epoch": 3,
+  "global_step": 484,
+  "len_train_dataloader": 121,
+  "training_strategy": "lora",
+  "model_order": [
+    "base",
+    "video",
+    "audio"
+  ],
+  "saved_at_unix": 1758161837.8391721
+}

video_preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,55 @@

+{
+  "chunk_length": 300,
+  "crop_size": null,
+  "data_format": "channels_first",
+  "default_to_square": true,
+  "device": null,
+  "dither": 0.0,
+  "do_center_crop": null,
+  "do_convert_rgb": true,
+  "do_normalize": true,
+  "do_pad": null,
+  "do_rescale": true,
+  "do_resize": true,
+  "do_sample_frames": false,
+  "feature_extractor_type": "WhisperFeatureExtractor",
+  "feature_size": 128,
+  "fps": null,
+  "hop_length": 160,
+  "image_mean": [
+    0.48145466,
+    0.4578275,
+    0.40821073
+  ],
+  "image_std": [
+    0.26862954,
+    0.26130258,
+    0.27577711
+  ],
+  "input_data_format": null,
+  "max_frames": 768,
+  "max_pixels": 12845056,
+  "merge_size": 2,
+  "min_frames": 4,
+  "min_pixels": 3136,
+  "n_fft": 400,
+  "n_samples": 4800000,
+  "nb_max_frames": 30000,
+  "num_frames": null,
+  "padding_side": "right",
+  "padding_value": 0.0,
+  "patch_size": 14,
+  "processor_class": "Qwen2_5OmniProcessor",
+  "resample": 3,
+  "rescale_factor": 0.00392156862745098,
+  "return_attention_mask": true,
+  "sampling_rate": 16000,
+  "size": {
+    "longest_edge": 12845056,
+    "shortest_edge": 3136
+  },
+  "size_divisor": null,
+  "temporal_patch_size": 2,
+  "video_metadata": null,
+  "video_processor_type": "Qwen2VLVideoProcessor"
+}

vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff