zhifeixie commited on 3 days ago

Commit

72ee2c2

verified ·

1 Parent(s): 02abc58

Add files using upload-large-folder tool

Browse files

This view is limited to 50 files because it contains too many changes. See raw diff

Files changed (50) hide show

.gitattributes +2 -0
lora/lora-stage1/README.md +207 -0
lora/lora-stage1/adapter_config.json +38 -0
lora/lora-stage1/adapter_model.safetensors +3 -0
lora/lora-stage1/added_tokens.json +64 -0
lora/lora-stage1/base_model.txt +1 -0
lora/lora-stage1/chat_template.jinja +31 -0
lora/lora-stage1/chat_template.json +1 -0
lora/lora-stage1/config.json +221 -0
lora/lora-stage1/generation_config.json +7 -0
lora/lora-stage1/merges.txt +0 -0
lora/lora-stage1/preprocessor_config.json +14 -0
lora/lora-stage1/rng_state_0.pth +3 -0
lora/lora-stage1/special_tokens_map.json +44 -0
lora/lora-stage1/tokenizer.json +3 -0
lora/lora-stage1/tokenizer_config.json +549 -0
lora/lora-stage1/trainer_state.json +774 -0
lora/lora-stage1/vocab.json +0 -0
lora/lora-stage2/README.md +207 -0
lora/lora-stage2/adapter_config.json +38 -0
lora/lora-stage2/adapter_model.safetensors +3 -0
lora/lora-stage2/added_tokens.json +64 -0
lora/lora-stage2/base_model.txt +1 -0
lora/lora-stage2/chat_template.jinja +31 -0
lora/lora-stage2/chat_template.json +1 -0
lora/lora-stage2/config.json +221 -0
lora/lora-stage2/generation_config.json +7 -0
lora/lora-stage2/merged_from_lora.txt +1 -0
lora/lora-stage2/merges.txt +0 -0
lora/lora-stage2/optimizer.pt +3 -0
lora/lora-stage2/preprocessor_config.json +14 -0
lora/lora-stage2/rng_state_0.pth +3 -0
lora/lora-stage2/rng_state_1.pth +3 -0
lora/lora-stage2/scheduler.pt +3 -0
lora/lora-stage2/special_tokens_map.json +44 -0
lora/lora-stage2/tokenizer.json +3 -0
lora/lora-stage2/tokenizer_config.json +549 -0
lora/lora-stage2/trainer_state.json +0 -0
lora/lora-stage2/vocab.json +0 -0
lora/lora-stage3/README.md +207 -0
lora/lora-stage3/adapter_config.json +38 -0
lora/lora-stage3/adapter_model.safetensors +3 -0
lora/lora-stage3/additional_config.json +1 -0
lora/lora-stage3/args.json +502 -0
lora/lora-stage3/optimizer.pt +3 -0
lora/lora-stage3/rng_state_0.pth +3 -0
lora/lora-stage3/rng_state_1.pth +3 -0
lora/lora-stage3/rng_state_2.pth +3 -0
lora/lora-stage3/scheduler.pt +3 -0
lora/lora-stage3/trainer_state.json +0 -0

.gitattributes CHANGED Viewed

@@ -33,3 +33,5 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text

 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
+lora/lora-stage1/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+lora/lora-stage2/tokenizer.json filter=lfs diff=lfs merge=lfs -text

lora/lora-stage1/README.md ADDED Viewed

	@@ -0,0 +1,207 @@

+---
+base_model: ''
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- 'base_model:adapter:'
+- lora
+- transformers
+---
+# Model Card for Model ID
+<!-- Provide a quick summary of what the model is/does. -->
+## Model Details
+### Model Description
+<!-- Provide a longer summary of what this model is. -->
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+### Model Sources [optional]
+<!-- Provide the basic links for the model. -->
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+## Uses
+<!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
+### Direct Use
+<!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
+[More Information Needed]
+### Downstream Use [optional]
+<!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
+[More Information Needed]
+### Out-of-Scope Use
+<!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
+[More Information Needed]
+## Bias, Risks, and Limitations
+<!-- This section is meant to convey both technical and sociotechnical limitations. -->
+[More Information Needed]
+### Recommendations
+<!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+## How to Get Started with the Model
+Use the code below to get started with the model.
+[More Information Needed]
+## Training Details
+### Training Data
+<!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
+[More Information Needed]
+### Training Procedure
+<!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
+#### Preprocessing [optional]
+[More Information Needed]
+#### Training Hyperparameters
+- **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
+#### Speeds, Sizes, Times [optional]
+<!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
+[More Information Needed]
+## Evaluation
+<!-- This section describes the evaluation protocols and provides the results. -->
+### Testing Data, Factors & Metrics
+#### Testing Data
+<!-- This should link to a Dataset Card if possible. -->
+[More Information Needed]
+#### Factors
+<!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
+[More Information Needed]
+#### Metrics
+<!-- These are the evaluation metrics being used, ideally with a description of why. -->
+[More Information Needed]
+### Results
+[More Information Needed]
+#### Summary
+## Model Examination [optional]
+<!-- Relevant interpretability work for the model goes here -->
+[More Information Needed]
+## Environmental Impact
+<!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+## Technical Specifications [optional]
+### Model Architecture and Objective
+[More Information Needed]
+### Compute Infrastructure
+[More Information Needed]
+#### Hardware
+[More Information Needed]
+#### Software
+[More Information Needed]
+## Citation [optional]
+<!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
+**BibTeX:**
+[More Information Needed]
+**APA:**
+[More Information Needed]
+## Glossary [optional]
+<!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
+[More Information Needed]
+## More Information [optional]
+[More Information Needed]
+## Model Card Authors [optional]
+[More Information Needed]
+## Model Card Contact
+[More Information Needed]
+### Framework versions
+- PEFT 0.18.1

lora/lora-stage1/adapter_config.json ADDED Viewed

	@@ -0,0 +1,38 @@

+{
+  "alora_invocation_tokens": null,
+  "alpha_pattern": {},
+  "arrow_config": null,
+  "auto_mapping": null,
+  "base_model_name_or_path": "",
+  "bias": "none",
+  "corda_config": null,
+  "ensure_weight_tying": false,
+  "eva_config": null,
+  "exclude_modules": null,
+  "fan_in_fan_out": false,
+  "inference_mode": true,
+  "init_lora_weights": true,
+  "layer_replication": null,
+  "layers_pattern": null,
+  "layers_to_transform": null,
+  "loftq_config": {},
+  "lora_alpha": 16,
+  "lora_bias": false,
+  "lora_dropout": 0.05,
+  "megatron_config": null,
+  "megatron_core": "megatron.core",
+  "modules_to_save": null,
+  "peft_type": "LORA",
+  "peft_version": "0.18.1",
+  "qalora_group_size": 16,
+  "r": 8,
+  "rank_pattern": {},
+  "revision": null,
+  "target_modules": "^(audio_tower\\.(conv_out|proj1|proj2)$|audio_tower\\.layers\\.(20|21|22|23)\\..*\\.(q_proj|k_proj|v_proj|out_proj|fc1|fc2)$)",
+  "target_parameters": null,
+  "task_type": "CAUSAL_LM",
+  "trainable_token_indices": null,
+  "use_dora": false,
+  "use_qalora": false,
+  "use_rslora": false
+}

lora/lora-stage1/adapter_model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:31052a993cbb582a250886db7dfcc327ab86ee8adc5229882bd48227b892c752
+size 1496072

lora/lora-stage1/added_tokens.json ADDED Viewed

	@@ -0,0 +1,64 @@

+{
+  "</think>": 151668,
+  "</tool_call>": 151658,
+  "</tool_response>": 151666,
+  "<asr_text>": 151704,
+  "<blank10>": 151686,
+  "<blank11>": 151687,
+  "<blank12>": 151688,
+  "<blank13>": 151689,
+  "<blank14>": 151690,
+  "<blank15>": 151691,
+  "<blank16>": 151692,
+  "<blank17>": 151693,
+  "<blank18>": 151694,
+  "<blank19>": 151695,
+  "<blank1>": 151677,
+  "<blank20>": 151696,
+  "<blank21>": 151697,
+  "<blank22>": 151698,
+  "<blank23>": 151699,
+  "<blank24>": 151700,
+  "<blank25>": 151701,
+  "<blank26>": 151702,
+  "<blank27>": 151703,
+  "<blank2>": 151678,
+  "<blank3>": 151679,
+  "<blank4>": 151680,
+  "<blank5>": 151681,
+  "<blank6>": 151682,
+  "<blank7>": 151683,
+  "<blank8>": 151684,
+  "<blank9>": 151685,
+  "<non_speech>": 151675,
+  "<think>": 151667,
+  "<tool_call>": 151657,
+  "<tool_response>": 151665,
+  "<tts_pad>": 151671,
+  "<tts_text_bos>": 151672,
+  "<tts_text_bos_single>": 151674,
+  "<tts_text_eod>": 151673,
+  "<|audio_end|>": 151670,
+  "<|audio_pad|>": 151676,
+  "<|audio_start|>": 151669,
+  "<|box_end|>": 151649,
+  "<|box_start|>": 151648,
+  "<|endoftext|>": 151643,
+  "<|file_sep|>": 151664,
+  "<|fim_middle|>": 151660,
+  "<|fim_pad|>": 151662,
+  "<|fim_prefix|>": 151659,
+  "<|fim_suffix|>": 151661,
+  "<|im_end|>": 151645,
+  "<|im_start|>": 151644,
+  "<|image_pad|>": 151655,
+  "<|object_ref_end|>": 151647,
+  "<|object_ref_start|>": 151646,
+  "<|quad_end|>": 151651,
+  "<|quad_start|>": 151650,
+  "<|repo_name|>": 151663,
+  "<|video_pad|>": 151656,
+  "<|vision_end|>": 151653,
+  "<|vision_pad|>": 151654,
+  "<|vision_start|>": 151652
+}

lora/lora-stage1/base_model.txt ADDED Viewed

	@@ -0,0 +1 @@


1	+ /data/haobin/pky_train/qwen3/Qwen3-ASR-1.7B

lora/lora-stage1/chat_template.jinja ADDED Viewed

	@@ -0,0 +1,31 @@

+{%- set ns = namespace(system_text="") -%}
+{%- for m in messages -%}
+  {%- if m.role == 'system' -%}
+    {%- if m.content is string -%}
+      {%- set ns.system_text = ns.system_text + m.content -%}
+    {%- else -%}
+      {%- for c in m.content -%}
+        {%- if c.type == 'text' and (c.text is defined) -%}
+          {%- set ns.system_text = ns.system_text + c.text -%}
+        {%- endif -%}
+      {%- endfor -%}
+    {%- endif -%}
+  {%- endif -%}
+{%- endfor -%}
+{%- set ns2 = namespace(audio_tokens="") -%}
+{%- for m in messages -%}
+  {%- if m.content is not string -%}
+    {%- for c in m.content -%}
+      {%- if c.type == 'audio' or ('audio' in c) or ('audio_url' in c) -%}
+        {%- set ns2.audio_tokens = ns2.audio_tokens + "<|audio_start|><|audio_pad|><|audio_end|>" -%}
+      {%- endif -%}
+    {%- endfor -%}
+  {%- endif -%}
+{%- endfor -%}
+{{- '<|im_start|>system\n' + (ns.system_text if ns.system_text is string else '') + '<|im_end|>\n' -}}
+{{- '<|im_start|>user\n' + ns2.audio_tokens + '<|im_end|>\n' -}}
+{%- if add_generation_prompt -%}
+{{- '<|im_start|>assistant\n' -}}
+{%- endif -%}

lora/lora-stage1/chat_template.json ADDED Viewed

	@@ -0,0 +1 @@

+ {"chat_template": "{%- set ns = namespace(system_text=\"\") -%}\n{%- for m in messages -%}\n {%- if m.role == 'system' -%}\n {%- if m.content is string -%}\n {%- set ns.system_text = ns.system_text + m.content -%}\n {%- else -%}\n {%- for c in m.content -%}\n {%- if c.type == 'text' and (c.text is defined) -%}\n {%- set ns.system_text = ns.system_text + c.text -%}\n {%- endif -%}\n {%- endfor -%}\n {%- endif -%}\n {%- endif -%}\n{%- endfor -%}\n\n{%- set ns2 = namespace(audio_tokens=\"\") -%}\n{%- for m in messages -%}\n {%- if m.content is not string -%}\n {%- for c in m.content -%}\n {%- if c.type == 'audio' or ('audio' in c) or ('audio_url' in c) -%}\n {%- set ns2.audio_tokens = ns2.audio_tokens + \"<|audio_start|><|audio_pad|><|audio_end|>\" -%}\n {%- endif -%}\n {%- endfor -%}\n {%- endif -%}\n{%- endfor -%}\n\n{{- '<|im_start|>system\\n' + (ns.system_text if ns.system_text is string else '') + '<|im_end|>\\n' -}}\n{{- '<|im_start|>user\\n' + ns2.audio_tokens + '<|im_end|>\\n' -}}\n{%- if add_generation_prompt -%}\n{{- '<|im_start|>assistant\\n' -}}\n{%- endif -%}"}

lora/lora-stage1/config.json ADDED Viewed

	@@ -0,0 +1,221 @@

+{
+    "architectures": [
+      "Qwen3ASRForConditionalGeneration"
+    ],
+    "model_type": "qwen3_asr",
+    "support_languages": [
+      "Chinese",
+      "English",
+      "Cantonese",
+      "Arabic",
+      "German",
+      "French",
+      "Spanish",
+      "Portuguese",
+      "Indonesian",
+      "Italian",
+      "Korean",
+      "Russian",
+      "Thai",
+      "Vietnamese",
+      "Japanese",
+      "Turkish",
+      "Hindi",
+      "Malay",
+      "Dutch",
+      "Swedish",
+      "Danish",
+      "Finnish",
+      "Polish",
+      "Czech",
+      "Filipino",
+      "Persian",
+      "Greek",
+      "Romanian",
+      "Hungarian",
+      "Macedonian"
+    ],
+    "thinker_config": {
+      "model_type": "qwen3_asr",
+      "architectures": [
+        "Qwen3ASRForConditionalGeneration"
+      ],
+      "audio_config": {
+        "_name_or_path": "",
+        "activation_dropout": 0,
+        "activation_function": "gelu",
+        "add_cross_attention": false,
+        "architectures": null,
+        "attention_dropout": 0,
+        "bad_words_ids": null,
+        "begin_suppress_tokens": null,
+        "bos_token_id": null,
+        "chunk_size_feed_forward": 0,
+        "conv_chunksize": 500,
+        "cross_attention_hidden_size": null,
+        "d_model": 1024,
+        "decoder_start_token_id": null,
+        "diversity_penalty": 0.0,
+        "do_sample": false,
+        "downsample_hidden_size": 480,
+        "dropout": 0,
+        "dtype": null,
+        "early_stopping": false,
+        "encoder_attention_heads": 16,
+        "encoder_ffn_dim": 4096,
+        "encoder_layers": 24,
+        "encoder_no_repeat_ngram_size": 0,
+        "eos_token_id": null,
+        "exponential_decay_length_penalty": null,
+        "finetuning_task": null,
+        "forced_bos_token_id": null,
+        "forced_eos_token_id": null,
+        "id2label": {
+          "0": "LABEL_0",
+          "1": "LABEL_1"
+        },
+        "initializer_range": 0.02,
+        "is_decoder": false,
+        "is_encoder_decoder": false,
+        "label2id": {
+          "LABEL_0": 0,
+          "LABEL_1": 1
+        },
+        "length_penalty": 1.0,
+        "max_length": 20,
+        "max_source_positions": 1500,
+        "min_length": 0,
+        "model_type": "qwen3_asr_audio_encoder",
+        "n_window": 50,
+        "n_window_infer": 800,
+        "no_repeat_ngram_size": 0,
+        "num_beam_groups": 1,
+        "num_beams": 1,
+        "num_hidden_layers": 24,
+        "num_mel_bins": 128,
+        "num_return_sequences": 1,
+        "output_attentions": false,
+        "output_dim": 2048,
+        "output_hidden_states": false,
+        "output_scores": false,
+        "pad_token_id": null,
+        "prefix": null,
+        "problem_type": null,
+        "pruned_heads": {},
+        "remove_invalid_values": false,
+        "repetition_penalty": 1.0,
+        "return_dict": true,
+        "return_dict_in_generate": false,
+        "scale_embedding": false,
+        "sep_token_id": null,
+        "suppress_tokens": null,
+        "task_specific_params": null,
+        "temperature": 1.0,
+        "tf_legacy_loss": false,
+        "tie_encoder_decoder": false,
+        "tie_word_embeddings": true,
+        "tokenizer_class": null,
+        "top_k": 50,
+        "top_p": 1.0,
+        "torchscript": false,
+        "typical_p": 1.0,
+        "use_bfloat16": false
+      },
+      "audio_end_token_id": 151670,
+      "audio_start_token_id": 151669,
+      "audio_token_id": 151676,
+      "dtype": "bfloat16",
+      "initializer_range": 0.02,
+      "text_config": {
+        "_name_or_path": "",
+        "add_cross_attention": false,
+        "architectures": null,
+        "attention_bias": false,
+        "attention_dropout": 0.0,
+        "bad_words_ids": null,
+        "begin_suppress_tokens": null,
+        "bos_token_id": null,
+        "chunk_size_feed_forward": 0,
+        "cross_attention_hidden_size": null,
+        "decoder_start_token_id": null,
+        "diversity_penalty": 0.0,
+        "do_sample": false,
+        "dtype": null,
+        "early_stopping": false,
+        "encoder_no_repeat_ngram_size": 0,
+        "eos_token_id": null,
+        "exponential_decay_length_penalty": null,
+        "finetuning_task": null,
+        "forced_bos_token_id": null,
+        "forced_eos_token_id": null,
+        "head_dim": 128,
+        "hidden_act": "silu",
+        "hidden_size": 2048,
+        "id2label": {
+          "0": "LABEL_0",
+          "1": "LABEL_1"
+        },
+        "initializer_range": 0.02,
+        "intermediate_size": 6144,
+        "is_decoder": false,
+        "is_encoder_decoder": false,
+        "label2id": {
+          "LABEL_0": 0,
+          "LABEL_1": 1
+        },
+        "length_penalty": 1.0,
+        "max_length": 20,
+        "max_position_embeddings": 65536,
+        "min_length": 0,
+        "model_type": "qwen3",
+        "no_repeat_ngram_size": 0,
+        "num_attention_heads": 16,
+        "num_beam_groups": 1,
+        "num_beams": 1,
+        "num_hidden_layers": 28,
+        "num_key_value_heads": 8,
+        "num_return_sequences": 1,
+        "output_attentions": false,
+        "output_hidden_states": false,
+        "output_scores": false,
+        "pad_token_id": null,
+        "prefix": null,
+        "problem_type": null,
+        "pruned_heads": {},
+        "remove_invalid_values": false,
+        "repetition_penalty": 1.0,
+        "return_dict": true,
+        "return_dict_in_generate": false,
+        "rms_norm_eps": 1e-06,
+        "rope_scaling": {
+          "interleaved": true,
+          "mrope_interleaved": true,
+          "mrope_section": [
+            24,
+            20,
+            20
+          ],
+          "rope_type": "default",
+          "type": "default"
+        },
+        "rope_theta": 1000000,
+        "sep_token_id": null,
+        "suppress_tokens": null,
+        "task_specific_params": null,
+        "temperature": 1.0,
+        "tf_legacy_loss": false,
+        "tie_encoder_decoder": false,
+        "tie_word_embeddings": true,
+        "tokenizer_class": null,
+        "top_k": 50,
+        "top_p": 1.0,
+        "torchscript": false,
+        "typical_p": 1.0,
+        "use_bfloat16": false,
+        "use_cache": true,
+        "vocab_size": 151936
+      }
+    },
+    "transformers_version": "4.57.6"
+  }

lora/lora-stage1/generation_config.json ADDED Viewed

	@@ -0,0 +1,7 @@

+{
+  "_from_model_config": true,
+  "eos_token_id": [151643,151645],
+  "pad_token_id": 151643,
+  "do_sample": false,
+  "temperature": 0.000001
+}

lora/lora-stage1/merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

lora/lora-stage1/preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,14 @@

+{
+  "chunk_length": 30,
+  "dither": 0.0,
+  "feature_extractor_type": "WhisperFeatureExtractor",
+  "feature_size": 128,
+  "hop_length": 160,
+  "n_fft": 400,
+  "n_samples": 480000,
+  "nb_max_frames": 3000,
+  "padding_side": "right",
+  "padding_value": 0.0,
+  "processor_class": "Qwen3ASRProcessor",
+  "return_attention_mask": true
+}

lora/lora-stage1/rng_state_0.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:916059f3d5e18a65741db0b5dc2209e8c6aad0736bace4b346dacc3a0ed5408c
+size 14917

lora/lora-stage1/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,44 @@

+{
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|object_ref_start|>",
+    "<|object_ref_end|>",
+    "<|box_start|>",
+    "<|box_end|>",
+    "<|quad_start|>",
+    "<|quad_end|>",
+    "<|vision_start|>",
+    "<|vision_end|>",
+    "<|vision_pad|>",
+    "<|image_pad|>",
+    "<|video_pad|>",
+    "<|audio_start|>",
+    "<|audio_end|>",
+    "<tts_pad>",
+    "<tts_text_bos>",
+    "<tts_text_bos_single>",
+    "<|audio_pad|>"
+  ],
+  "audio_bos_token": "<|audio_start|>",
+  "audio_eos_token": "<|audio_end|>",
+  "audio_token": "<|audio_pad|>",
+  "eos_token": {
+    "content": "<|im_end|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "image_token": "<|image_pad|>",
+  "pad_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "video_token": "<|video_pad|>",
+  "vision_bos_token": "<|vision_start|>",
+  "vision_eos_token": "<|vision_end|>"
+}

lora/lora-stage1/tokenizer.json ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:0499602714160467f2d68b910651d6216020689f1e016be87a2d0019ee3baeab
+size 11429499

lora/lora-stage1/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,549 @@

+{
+  "add_bos_token": false,
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "151643": {
+      "content": "<|endoftext|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151644": {
+      "content": "<|im_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151645": {
+      "content": "<|im_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151646": {
+      "content": "<|object_ref_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151647": {
+      "content": "<|object_ref_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151648": {
+      "content": "<|box_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151649": {
+      "content": "<|box_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151650": {
+      "content": "<|quad_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151651": {
+      "content": "<|quad_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151652": {
+      "content": "<|vision_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151653": {
+      "content": "<|vision_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151654": {
+      "content": "<|vision_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151655": {
+      "content": "<|image_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151656": {
+      "content": "<|video_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151657": {
+      "content": "<tool_call>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151658": {
+      "content": "</tool_call>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151659": {
+      "content": "<|fim_prefix|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151660": {
+      "content": "<|fim_middle|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151661": {
+      "content": "<|fim_suffix|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151662": {
+      "content": "<|fim_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151663": {
+      "content": "<|repo_name|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151664": {
+      "content": "<|file_sep|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151665": {
+      "content": "<tool_response>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151666": {
+      "content": "</tool_response>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151667": {
+      "content": "<think>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151668": {
+      "content": "</think>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151669": {
+      "content": "<|audio_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151670": {
+      "content": "<|audio_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151671": {
+      "content": "<tts_pad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151672": {
+      "content": "<tts_text_bos>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151673": {
+      "content": "<tts_text_eod>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151674": {
+      "content": "<tts_text_bos_single>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151675": {
+      "content": "<non_speech>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151676": {
+      "content": "<|audio_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151677": {
+      "content": "<blank1>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151678": {
+      "content": "<blank2>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151679": {
+      "content": "<blank3>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151680": {
+      "content": "<blank4>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151681": {
+      "content": "<blank5>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151682": {
+      "content": "<blank6>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151683": {
+      "content": "<blank7>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151684": {
+      "content": "<blank8>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151685": {
+      "content": "<blank9>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151686": {
+      "content": "<blank10>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151687": {
+      "content": "<blank11>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151688": {
+      "content": "<blank12>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151689": {
+      "content": "<blank13>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151690": {
+      "content": "<blank14>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151691": {
+      "content": "<blank15>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151692": {
+      "content": "<blank16>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151693": {
+      "content": "<blank17>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151694": {
+      "content": "<blank18>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151695": {
+      "content": "<blank19>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151696": {
+      "content": "<blank20>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151697": {
+      "content": "<blank21>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151698": {
+      "content": "<blank22>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151699": {
+      "content": "<blank23>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151700": {
+      "content": "<blank24>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151701": {
+      "content": "<blank25>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151702": {
+      "content": "<blank26>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151703": {
+      "content": "<blank27>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151704": {
+      "content": "<asr_text>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    }
+  },
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|object_ref_start|>",
+    "<|object_ref_end|>",
+    "<|box_start|>",
+    "<|box_end|>",
+    "<|quad_start|>",
+    "<|quad_end|>",
+    "<|vision_start|>",
+    "<|vision_end|>",
+    "<|vision_pad|>",
+    "<|image_pad|>",
+    "<|video_pad|>",
+    "<|audio_start|>",
+    "<|audio_end|>",
+    "<tts_pad>",
+    "<tts_text_bos>",
+    "<tts_text_bos_single>",
+    "<|audio_pad|>"
+  ],
+  "audio_bos_token": "<|audio_start|>",
+  "audio_eos_token": "<|audio_end|>",
+  "audio_token": "<|audio_pad|>",
+  "bos_token": null,
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<|im_end|>",
+  "errors": "replace",
+  "extra_special_tokens": {
+    "audio_bos_token": "<|audio_start|>",
+    "audio_eos_token": "<|audio_end|>",
+    "audio_token": "<|audio_pad|>",
+    "image_token": "<|image_pad|>",
+    "video_token": "<|video_pad|>",
+    "vision_bos_token": "<|vision_start|>",
+    "vision_eos_token": "<|vision_end|>"
+  },
+  "image_token": "<|image_pad|>",
+  "model_max_length": 131072,
+  "pad_token": "<|endoftext|>",
+  "processor_class": "Qwen3ASRProcessor",
+  "split_special_tokens": false,
+  "tokenizer_class": "Qwen2Tokenizer",
+  "unk_token": null,
+  "video_token": "<|video_pad|>",
+  "vision_bos_token": "<|vision_start|>",
+  "vision_eos_token": "<|vision_end|>"
+}

lora/lora-stage1/trainer_state.json ADDED Viewed

	@@ -0,0 +1,774 @@

+{
+  "best_global_step": null,
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 0.29335191228777824,
+  "eval_steps": 200,
+  "global_step": 1000,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 0.002933519122877782,
+      "grad_norm": 31.058988571166992,
+      "learning_rate": 2.6392961876832844e-08,
+      "loss": 222.2233,
+      "step": 10
+    },
+    {
+      "epoch": 0.005867038245755564,
+      "grad_norm": 29.318532943725586,
+      "learning_rate": 5.571847507331378e-08,
+      "loss": 223.4508,
+      "step": 20
+    },
+    {
+      "epoch": 0.008800557368633347,
+      "grad_norm": 31.036029815673828,
+      "learning_rate": 8.504398826979471e-08,
+      "loss": 223.5497,
+      "step": 30
+    },
+    {
+      "epoch": 0.011734076491511128,
+      "grad_norm": 31.80939483642578,
+      "learning_rate": 1.1436950146627565e-07,
+      "loss": 218.2694,
+      "step": 40
+    },
+    {
+      "epoch": 0.014667595614388912,
+      "grad_norm": 32.80522918701172,
+      "learning_rate": 1.436950146627566e-07,
+      "loss": 219.2423,
+      "step": 50
+    },
+    {
+      "epoch": 0.017601114737266693,
+      "grad_norm": 32.718772888183594,
+      "learning_rate": 1.7302052785923753e-07,
+      "loss": 224.9209,
+      "step": 60
+    },
+    {
+      "epoch": 0.020534633860144477,
+      "grad_norm": 30.853660583496094,
+      "learning_rate": 2.0234604105571846e-07,
+      "loss": 220.9806,
+      "step": 70
+    },
+    {
+      "epoch": 0.023468152983022256,
+      "grad_norm": 31.83987045288086,
+      "learning_rate": 2.3167155425219938e-07,
+      "loss": 221.7758,
+      "step": 80
+    },
+    {
+      "epoch": 0.02640167210590004,
+      "grad_norm": 33.82211685180664,
+      "learning_rate": 2.609970674486803e-07,
+      "loss": 220.2104,
+      "step": 90
+    },
+    {
+      "epoch": 0.029335191228777823,
+      "grad_norm": 39.342655181884766,
+      "learning_rate": 2.903225806451613e-07,
+      "loss": 223.5162,
+      "step": 100
+    },
+    {
+      "epoch": 0.032268710351655606,
+      "grad_norm": 32.44097900390625,
+      "learning_rate": 3.196480938416422e-07,
+      "loss": 222.4887,
+      "step": 110
+    },
+    {
+      "epoch": 0.035202229474533386,
+      "grad_norm": 30.906185150146484,
+      "learning_rate": 3.489736070381232e-07,
+      "loss": 221.0162,
+      "step": 120
+    },
+    {
+      "epoch": 0.038135748597411166,
+      "grad_norm": 30.318588256835938,
+      "learning_rate": 3.7829912023460407e-07,
+      "loss": 219.1895,
+      "step": 130
+    },
+    {
+      "epoch": 0.04106926772028895,
+      "grad_norm": 33.13260269165039,
+      "learning_rate": 4.0762463343108505e-07,
+      "loss": 219.1354,
+      "step": 140
+    },
+    {
+      "epoch": 0.04400278684316673,
+      "grad_norm": 32.98201370239258,
+      "learning_rate": 4.36950146627566e-07,
+      "loss": 221.9035,
+      "step": 150
+    },
+    {
+      "epoch": 0.04693630596604451,
+      "grad_norm": 30.733919143676758,
+      "learning_rate": 4.6627565982404685e-07,
+      "loss": 219.0109,
+      "step": 160
+    },
+    {
+      "epoch": 0.0498698250889223,
+      "grad_norm": 35.68417739868164,
+      "learning_rate": 4.956011730205278e-07,
+      "loss": 222.1004,
+      "step": 170
+    },
+    {
+      "epoch": 0.05280334421180008,
+      "grad_norm": 34.876121520996094,
+      "learning_rate": 5.249266862170088e-07,
+      "loss": 220.137,
+      "step": 180
+    },
+    {
+      "epoch": 0.055736863334677866,
+      "grad_norm": 33.82151412963867,
+      "learning_rate": 5.542521994134897e-07,
+      "loss": 224.7452,
+      "step": 190
+    },
+    {
+      "epoch": 0.058670382457555646,
+      "grad_norm": 36.70476531982422,
+      "learning_rate": 5.835777126099707e-07,
+      "loss": 219.8298,
+      "step": 200
+    },
+    {
+      "epoch": 0.058670382457555646,
+      "eval_loss": 24.500732421875,
+      "eval_runtime": 98.9198,
+      "eval_samples_per_second": 98.019,
+      "eval_steps_per_second": 6.126,
+      "step": 200
+    },
+    {
+      "epoch": 0.061603901580433426,
+      "grad_norm": 34.49006652832031,
+      "learning_rate": 6.129032258064516e-07,
+      "loss": 223.5638,
+      "step": 210
+    },
+    {
+      "epoch": 0.06453742070331121,
+      "grad_norm": 32.312313079833984,
+      "learning_rate": 6.422287390029325e-07,
+      "loss": 225.3921,
+      "step": 220
+    },
+    {
+      "epoch": 0.06747093982618899,
+      "grad_norm": 33.46302032470703,
+      "learning_rate": 6.715542521994134e-07,
+      "loss": 219.3619,
+      "step": 230
+    },
+    {
+      "epoch": 0.07040445894906677,
+      "grad_norm": 47.695858001708984,
+      "learning_rate": 7.008797653958944e-07,
+      "loss": 221.6162,
+      "step": 240
+    },
+    {
+      "epoch": 0.07333797807194456,
+      "grad_norm": 36.99955368041992,
+      "learning_rate": 7.302052785923753e-07,
+      "loss": 224.4357,
+      "step": 250
+    },
+    {
+      "epoch": 0.07627149719482233,
+      "grad_norm": 33.713096618652344,
+      "learning_rate": 7.595307917888563e-07,
+      "loss": 218.6644,
+      "step": 260
+    },
+    {
+      "epoch": 0.07920501631770012,
+      "grad_norm": 36.349666595458984,
+      "learning_rate": 7.888563049853372e-07,
+      "loss": 221.4383,
+      "step": 270
+    },
+    {
+      "epoch": 0.0821385354405779,
+      "grad_norm": 36.67658615112305,
+      "learning_rate": 8.181818181818182e-07,
+      "loss": 221.6365,
+      "step": 280
+    },
+    {
+      "epoch": 0.08507205456345568,
+      "grad_norm": 31.31206512451172,
+      "learning_rate": 8.475073313782992e-07,
+      "loss": 219.8238,
+      "step": 290
+    },
+    {
+      "epoch": 0.08800557368633347,
+      "grad_norm": 33.81391525268555,
+      "learning_rate": 8.7683284457478e-07,
+      "loss": 222.3335,
+      "step": 300
+    },
+    {
+      "epoch": 0.09093909280921125,
+      "grad_norm": 39.456138610839844,
+      "learning_rate": 9.061583577712609e-07,
+      "loss": 225.0574,
+      "step": 310
+    },
+    {
+      "epoch": 0.09387261193208903,
+      "grad_norm": 62.84433364868164,
+      "learning_rate": 9.354838709677418e-07,
+      "loss": 222.3193,
+      "step": 320
+    },
+    {
+      "epoch": 0.09680613105496681,
+      "grad_norm": 37.60541915893555,
+      "learning_rate": 9.648093841642228e-07,
+      "loss": 215.4717,
+      "step": 330
+    },
+    {
+      "epoch": 0.0997396501778446,
+      "grad_norm": 42.61164855957031,
+      "learning_rate": 9.941348973607037e-07,
+      "loss": 220.7702,
+      "step": 340
+    },
+    {
+      "epoch": 0.10267316930072237,
+      "grad_norm": 41.35678482055664,
+      "learning_rate": 9.987648602748184e-07,
+      "loss": 221.7166,
+      "step": 350
+    },
+    {
+      "epoch": 0.10560668842360016,
+      "grad_norm": 41.287208557128906,
+      "learning_rate": 9.972209356183417e-07,
+      "loss": 222.1618,
+      "step": 360
+    },
+    {
+      "epoch": 0.10854020754647795,
+      "grad_norm": 54.5716667175293,
+      "learning_rate": 9.956770109618649e-07,
+      "loss": 221.5792,
+      "step": 370
+    },
+    {
+      "epoch": 0.11147372666935573,
+      "grad_norm": 40.734012603759766,
+      "learning_rate": 9.941330863053883e-07,
+      "loss": 219.901,
+      "step": 380
+    },
+    {
+      "epoch": 0.1144072457922335,
+      "grad_norm": 43.457218170166016,
+      "learning_rate": 9.925891616489115e-07,
+      "loss": 223.7378,
+      "step": 390
+    },
+    {
+      "epoch": 0.11734076491511129,
+      "grad_norm": 42.917686462402344,
+      "learning_rate": 9.910452369924347e-07,
+      "loss": 222.8944,
+      "step": 400
+    },
+    {
+      "epoch": 0.11734076491511129,
+      "eval_loss": 24.425460815429688,
+      "eval_runtime": 94.8923,
+      "eval_samples_per_second": 102.179,
+      "eval_steps_per_second": 6.386,
+      "step": 400
+    },
+    {
+      "epoch": 0.12027428403798908,
+      "grad_norm": 39.965293884277344,
+      "learning_rate": 9.89501312335958e-07,
+      "loss": 220.1472,
+      "step": 410
+    },
+    {
+      "epoch": 0.12320780316086685,
+      "grad_norm": 45.19244384765625,
+      "learning_rate": 9.879573876794812e-07,
+      "loss": 224.2056,
+      "step": 420
+    },
+    {
+      "epoch": 0.12614132228374464,
+      "grad_norm": 41.27251434326172,
+      "learning_rate": 9.864134630230044e-07,
+      "loss": 217.4446,
+      "step": 430
+    },
+    {
+      "epoch": 0.12907484140662243,
+      "grad_norm": 49.71922302246094,
+      "learning_rate": 9.848695383665276e-07,
+      "loss": 220.2578,
+      "step": 440
+    },
+    {
+      "epoch": 0.1320083605295002,
+      "grad_norm": 65.56668853759766,
+      "learning_rate": 9.833256137100508e-07,
+      "loss": 221.2077,
+      "step": 450
+    },
+    {
+      "epoch": 0.13494187965237797,
+      "grad_norm": 41.73335266113281,
+      "learning_rate": 9.817816890535742e-07,
+      "loss": 219.8,
+      "step": 460
+    },
+    {
+      "epoch": 0.13787539877525576,
+      "grad_norm": 51.275718688964844,
+      "learning_rate": 9.802377643970974e-07,
+      "loss": 221.3817,
+      "step": 470
+    },
+    {
+      "epoch": 0.14080891789813355,
+      "grad_norm": 55.4876823425293,
+      "learning_rate": 9.786938397406207e-07,
+      "loss": 216.4269,
+      "step": 480
+    },
+    {
+      "epoch": 0.14374243702101133,
+      "grad_norm": 55.99393844604492,
+      "learning_rate": 9.771499150841439e-07,
+      "loss": 218.8694,
+      "step": 490
+    },
+    {
+      "epoch": 0.14667595614388912,
+      "grad_norm": 95.5741958618164,
+      "learning_rate": 9.75605990427667e-07,
+      "loss": 221.4839,
+      "step": 500
+    },
+    {
+      "epoch": 0.1496094752667669,
+      "grad_norm": 49.25442886352539,
+      "learning_rate": 9.740620657711903e-07,
+      "loss": 222.9515,
+      "step": 510
+    },
+    {
+      "epoch": 0.15254299438964466,
+      "grad_norm": 50.05457305908203,
+      "learning_rate": 9.725181411147135e-07,
+      "loss": 218.0743,
+      "step": 520
+    },
+    {
+      "epoch": 0.15547651351252245,
+      "grad_norm": 43.44709777832031,
+      "learning_rate": 9.709742164582367e-07,
+      "loss": 218.6208,
+      "step": 530
+    },
+    {
+      "epoch": 0.15841003263540024,
+      "grad_norm": 66.39103698730469,
+      "learning_rate": 9.694302918017602e-07,
+      "loss": 219.6833,
+      "step": 540
+    },
+    {
+      "epoch": 0.16134355175827803,
+      "grad_norm": 54.72968292236328,
+      "learning_rate": 9.678863671452832e-07,
+      "loss": 221.8852,
+      "step": 550
+    },
+    {
+      "epoch": 0.1642770708811558,
+      "grad_norm": 65.26374816894531,
+      "learning_rate": 9.663424424888064e-07,
+      "loss": 219.7626,
+      "step": 560
+    },
+    {
+      "epoch": 0.1672105900040336,
+      "grad_norm": 60.0925178527832,
+      "learning_rate": 9.647985178323296e-07,
+      "loss": 217.8218,
+      "step": 570
+    },
+    {
+      "epoch": 0.17014410912691136,
+      "grad_norm": 47.97535705566406,
+      "learning_rate": 9.63254593175853e-07,
+      "loss": 217.9315,
+      "step": 580
+    },
+    {
+      "epoch": 0.17307762824978914,
+      "grad_norm": 53.61656951904297,
+      "learning_rate": 9.617106685193762e-07,
+      "loss": 219.2269,
+      "step": 590
+    },
+    {
+      "epoch": 0.17601114737266693,
+      "grad_norm": 52.75293731689453,
+      "learning_rate": 9.601667438628995e-07,
+      "loss": 216.867,
+      "step": 600
+    },
+    {
+      "epoch": 0.17601114737266693,
+      "eval_loss": 24.240764617919922,
+      "eval_runtime": 97.5766,
+      "eval_samples_per_second": 99.368,
+      "eval_steps_per_second": 6.211,
+      "step": 600
+    },
+    {
+      "epoch": 0.17894466649554472,
+      "grad_norm": 59.573219299316406,
+      "learning_rate": 9.586228192064227e-07,
+      "loss": 213.2538,
+      "step": 610
+    },
+    {
+      "epoch": 0.1818781856184225,
+      "grad_norm": 113.46548461914062,
+      "learning_rate": 9.570788945499459e-07,
+      "loss": 218.0255,
+      "step": 620
+    },
+    {
+      "epoch": 0.1848117047413003,
+      "grad_norm": 119.12982177734375,
+      "learning_rate": 9.55534969893469e-07,
+      "loss": 216.9313,
+      "step": 630
+    },
+    {
+      "epoch": 0.18774522386417805,
+      "grad_norm": 54.008338928222656,
+      "learning_rate": 9.539910452369923e-07,
+      "loss": 220.8365,
+      "step": 640
+    },
+    {
+      "epoch": 0.19067874298705584,
+      "grad_norm": 59.56270217895508,
+      "learning_rate": 9.524471205805155e-07,
+      "loss": 218.8136,
+      "step": 650
+    },
+    {
+      "epoch": 0.19361226210993362,
+      "grad_norm": 52.067115783691406,
+      "learning_rate": 9.509031959240389e-07,
+      "loss": 220.6164,
+      "step": 660
+    },
+    {
+      "epoch": 0.1965457812328114,
+      "grad_norm": 60.61309051513672,
+      "learning_rate": 9.493592712675621e-07,
+      "loss": 217.9881,
+      "step": 670
+    },
+    {
+      "epoch": 0.1994793003556892,
+      "grad_norm": 49.88456726074219,
+      "learning_rate": 9.478153466110853e-07,
+      "loss": 217.0137,
+      "step": 680
+    },
+    {
+      "epoch": 0.20241281947856699,
+      "grad_norm": 49.28492736816406,
+      "learning_rate": 9.462714219546085e-07,
+      "loss": 212.2388,
+      "step": 690
+    },
+    {
+      "epoch": 0.20534633860144474,
+      "grad_norm": 55.44947814941406,
+      "learning_rate": 9.447274972981318e-07,
+      "loss": 221.7097,
+      "step": 700
+    },
+    {
+      "epoch": 0.20827985772432253,
+      "grad_norm": 47.7352409362793,
+      "learning_rate": 9.43183572641655e-07,
+      "loss": 217.8991,
+      "step": 710
+    },
+    {
+      "epoch": 0.21121337684720032,
+      "grad_norm": 56.91552734375,
+      "learning_rate": 9.416396479851782e-07,
+      "loss": 216.618,
+      "step": 720
+    },
+    {
+      "epoch": 0.2141468959700781,
+      "grad_norm": 50.68717575073242,
+      "learning_rate": 9.400957233287015e-07,
+      "loss": 217.7346,
+      "step": 730
+    },
+    {
+      "epoch": 0.2170804150929559,
+      "grad_norm": 75.52225494384766,
+      "learning_rate": 9.385517986722248e-07,
+      "loss": 215.9344,
+      "step": 740
+    },
+    {
+      "epoch": 0.22001393421583368,
+      "grad_norm": 74.4793472290039,
+      "learning_rate": 9.37007874015748e-07,
+      "loss": 222.193,
+      "step": 750
+    },
+    {
+      "epoch": 0.22294745333871147,
+      "grad_norm": 58.30630111694336,
+      "learning_rate": 9.354639493592712e-07,
+      "loss": 215.5639,
+      "step": 760
+    },
+    {
+      "epoch": 0.22588097246158922,
+      "grad_norm": 52.7680778503418,
+      "learning_rate": 9.339200247027944e-07,
+      "loss": 219.3169,
+      "step": 770
+    },
+    {
+      "epoch": 0.228814491584467,
+      "grad_norm": 51.10957717895508,
+      "learning_rate": 9.323761000463177e-07,
+      "loss": 213.7119,
+      "step": 780
+    },
+    {
+      "epoch": 0.2317480107073448,
+      "grad_norm": 96.71678161621094,
+      "learning_rate": 9.30832175389841e-07,
+      "loss": 216.2126,
+      "step": 790
+    },
+    {
+      "epoch": 0.23468152983022258,
+      "grad_norm": 59.496395111083984,
+      "learning_rate": 9.292882507333642e-07,
+      "loss": 220.2937,
+      "step": 800
+    },
+    {
+      "epoch": 0.23468152983022258,
+      "eval_loss": 24.050508499145508,
+      "eval_runtime": 98.6094,
+      "eval_samples_per_second": 98.327,
+      "eval_steps_per_second": 6.145,
+      "step": 800
+    },
+    {
+      "epoch": 0.23761504895310037,
+      "grad_norm": 115.57308959960938,
+      "learning_rate": 9.277443260768874e-07,
+      "loss": 214.0267,
+      "step": 810
+    },
+    {
+      "epoch": 0.24054856807597816,
+      "grad_norm": 58.29754638671875,
+      "learning_rate": 9.262004014204107e-07,
+      "loss": 219.2819,
+      "step": 820
+    },
+    {
+      "epoch": 0.24348208719885592,
+      "grad_norm": 137.19517517089844,
+      "learning_rate": 9.246564767639339e-07,
+      "loss": 217.8361,
+      "step": 830
+    },
+    {
+      "epoch": 0.2464156063217337,
+      "grad_norm": 62.34098434448242,
+      "learning_rate": 9.23112552107457e-07,
+      "loss": 217.4855,
+      "step": 840
+    },
+    {
+      "epoch": 0.2493491254446115,
+      "grad_norm": 57.445247650146484,
+      "learning_rate": 9.215686274509803e-07,
+      "loss": 217.8953,
+      "step": 850
+    },
+    {
+      "epoch": 0.2522826445674893,
+      "grad_norm": 61.09876251220703,
+      "learning_rate": 9.200247027945036e-07,
+      "loss": 215.2011,
+      "step": 860
+    },
+    {
+      "epoch": 0.25521616369036704,
+      "grad_norm": 59.176513671875,
+      "learning_rate": 9.184807781380268e-07,
+      "loss": 217.2304,
+      "step": 870
+    },
+    {
+      "epoch": 0.25814968281324485,
+      "grad_norm": 52.66059494018555,
+      "learning_rate": 9.1693685348155e-07,
+      "loss": 218.234,
+      "step": 880
+    },
+    {
+      "epoch": 0.2610832019361226,
+      "grad_norm": 98.39973449707031,
+      "learning_rate": 9.153929288250732e-07,
+      "loss": 214.297,
+      "step": 890
+    },
+    {
+      "epoch": 0.2640167210590004,
+      "grad_norm": 72.08065795898438,
+      "learning_rate": 9.138490041685965e-07,
+      "loss": 217.044,
+      "step": 900
+    },
+    {
+      "epoch": 0.2669502401818782,
+      "grad_norm": 59.712371826171875,
+      "learning_rate": 9.123050795121198e-07,
+      "loss": 215.2483,
+      "step": 910
+    },
+    {
+      "epoch": 0.26988375930475594,
+      "grad_norm": 64.43281555175781,
+      "learning_rate": 9.10761154855643e-07,
+      "loss": 211.9948,
+      "step": 920
+    },
+    {
+      "epoch": 0.27281727842763376,
+      "grad_norm": 61.78029251098633,
+      "learning_rate": 9.092172301991662e-07,
+      "loss": 217.2441,
+      "step": 930
+    },
+    {
+      "epoch": 0.2757507975505115,
+      "grad_norm": 68.14164733886719,
+      "learning_rate": 9.076733055426895e-07,
+      "loss": 214.7014,
+      "step": 940
+    },
+    {
+      "epoch": 0.27868431667338933,
+      "grad_norm": 61.65287399291992,
+      "learning_rate": 9.061293808862127e-07,
+      "loss": 212.859,
+      "step": 950
+    },
+    {
+      "epoch": 0.2816178357962671,
+      "grad_norm": 64.0514144897461,
+      "learning_rate": 9.045854562297359e-07,
+      "loss": 217.1946,
+      "step": 960
+    },
+    {
+      "epoch": 0.2845513549191449,
+      "grad_norm": 91.87364959716797,
+      "learning_rate": 9.030415315732592e-07,
+      "loss": 215.7542,
+      "step": 970
+    },
+    {
+      "epoch": 0.28748487404202266,
+      "grad_norm": 54.730316162109375,
+      "learning_rate": 9.014976069167825e-07,
+      "loss": 218.0408,
+      "step": 980
+    },
+    {
+      "epoch": 0.2904183931649004,
+      "grad_norm": 56.43712615966797,
+      "learning_rate": 8.999536822603057e-07,
+      "loss": 212.8671,
+      "step": 990
+    },
+    {
+      "epoch": 0.29335191228777824,
+      "grad_norm": 59.28590393066406,
+      "learning_rate": 8.984097576038289e-07,
+      "loss": 215.822,
+      "step": 1000
+    },
+    {
+      "epoch": 0.29335191228777824,
+      "eval_loss": 23.851858139038086,
+      "eval_runtime": 96.4448,
+      "eval_samples_per_second": 100.534,
+      "eval_steps_per_second": 6.283,
+      "step": 1000
+    }
+  ],
+  "logging_steps": 10,
+  "max_steps": 6818,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 2,
+  "save_steps": 200,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": true,
+        "should_training_stop": false
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 3.503007404654592e+17,
+  "train_batch_size": 8,
+  "trial_name": null,
+  "trial_params": null
+}

lora/lora-stage1/vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff

lora/lora-stage2/README.md ADDED Viewed

	@@ -0,0 +1,207 @@

+---
+base_model: ''
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- 'base_model:adapter:'
+- lora
+- transformers
+---
+# Model Card for Model ID
+<!-- Provide a quick summary of what the model is/does. -->
+## Model Details
+### Model Description
+<!-- Provide a longer summary of what this model is. -->
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+### Model Sources [optional]
+<!-- Provide the basic links for the model. -->
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+## Uses
+<!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
+### Direct Use
+<!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
+[More Information Needed]
+### Downstream Use [optional]
+<!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
+[More Information Needed]
+### Out-of-Scope Use
+<!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
+[More Information Needed]
+## Bias, Risks, and Limitations
+<!-- This section is meant to convey both technical and sociotechnical limitations. -->
+[More Information Needed]
+### Recommendations
+<!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+## How to Get Started with the Model
+Use the code below to get started with the model.
+[More Information Needed]
+## Training Details
+### Training Data
+<!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
+[More Information Needed]
+### Training Procedure
+<!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
+#### Preprocessing [optional]
+[More Information Needed]
+#### Training Hyperparameters
+- **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
+#### Speeds, Sizes, Times [optional]
+<!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
+[More Information Needed]
+## Evaluation
+<!-- This section describes the evaluation protocols and provides the results. -->
+### Testing Data, Factors & Metrics
+#### Testing Data
+<!-- This should link to a Dataset Card if possible. -->
+[More Information Needed]
+#### Factors
+<!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
+[More Information Needed]
+#### Metrics
+<!-- These are the evaluation metrics being used, ideally with a description of why. -->
+[More Information Needed]
+### Results
+[More Information Needed]
+#### Summary
+## Model Examination [optional]
+<!-- Relevant interpretability work for the model goes here -->
+[More Information Needed]
+## Environmental Impact
+<!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+## Technical Specifications [optional]
+### Model Architecture and Objective
+[More Information Needed]
+### Compute Infrastructure
+[More Information Needed]
+#### Hardware
+[More Information Needed]
+#### Software
+[More Information Needed]
+## Citation [optional]
+<!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
+**BibTeX:**
+[More Information Needed]
+**APA:**
+[More Information Needed]
+## Glossary [optional]
+<!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
+[More Information Needed]
+## More Information [optional]
+[More Information Needed]
+## Model Card Authors [optional]
+[More Information Needed]
+## Model Card Contact
+[More Information Needed]
+### Framework versions
+- PEFT 0.18.1

lora/lora-stage2/adapter_config.json ADDED Viewed

	@@ -0,0 +1,38 @@

+{
+  "alora_invocation_tokens": null,
+  "alpha_pattern": {},
+  "arrow_config": null,
+  "auto_mapping": null,
+  "base_model_name_or_path": "",
+  "bias": "none",
+  "corda_config": null,
+  "ensure_weight_tying": false,
+  "eva_config": null,
+  "exclude_modules": null,
+  "fan_in_fan_out": false,
+  "inference_mode": true,
+  "init_lora_weights": true,
+  "layer_replication": null,
+  "layers_pattern": null,
+  "layers_to_transform": null,
+  "loftq_config": {},
+  "lora_alpha": 16,
+  "lora_bias": false,
+  "lora_dropout": 0.05,
+  "megatron_config": null,
+  "megatron_core": "megatron.core",
+  "modules_to_save": null,
+  "peft_type": "LORA",
+  "peft_version": "0.18.1",
+  "qalora_group_size": 16,
+  "r": 8,
+  "rank_pattern": {},
+  "revision": null,
+  "target_modules": "^(audio_tower\\.(conv_out|proj1|proj2)$|audio_tower\\.layers\\.\\d+\\..*\\.(q_proj|k_proj|v_proj|out_proj|fc1|fc2)$|model\\.layers\\.\\d+\\..*\\.(q_proj|k_proj|v_proj|o_proj|gate_proj|up_proj|down_proj)$)",
+  "target_parameters": null,
+  "task_type": "CAUSAL_LM",
+  "trainable_token_indices": null,
+  "use_dora": false,
+  "use_qalora": false,
+  "use_rslora": false
+}

lora/lora-stage2/adapter_model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:dd4baa6a45645b280fdddb3c722186d149b5f64daab687300dba0c08373e3962
+size 41677888

lora/lora-stage2/added_tokens.json ADDED Viewed

	@@ -0,0 +1,64 @@

+{
+  "</think>": 151668,
+  "</tool_call>": 151658,
+  "</tool_response>": 151666,
+  "<asr_text>": 151704,
+  "<blank10>": 151686,
+  "<blank11>": 151687,
+  "<blank12>": 151688,
+  "<blank13>": 151689,
+  "<blank14>": 151690,
+  "<blank15>": 151691,
+  "<blank16>": 151692,
+  "<blank17>": 151693,
+  "<blank18>": 151694,
+  "<blank19>": 151695,
+  "<blank1>": 151677,
+  "<blank20>": 151696,
+  "<blank21>": 151697,
+  "<blank22>": 151698,
+  "<blank23>": 151699,
+  "<blank24>": 151700,
+  "<blank25>": 151701,
+  "<blank26>": 151702,
+  "<blank27>": 151703,
+  "<blank2>": 151678,
+  "<blank3>": 151679,
+  "<blank4>": 151680,
+  "<blank5>": 151681,
+  "<blank6>": 151682,
+  "<blank7>": 151683,
+  "<blank8>": 151684,
+  "<blank9>": 151685,
+  "<non_speech>": 151675,
+  "<think>": 151667,
+  "<tool_call>": 151657,
+  "<tool_response>": 151665,
+  "<tts_pad>": 151671,
+  "<tts_text_bos>": 151672,
+  "<tts_text_bos_single>": 151674,
+  "<tts_text_eod>": 151673,
+  "<|audio_end|>": 151670,
+  "<|audio_pad|>": 151676,
+  "<|audio_start|>": 151669,
+  "<|box_end|>": 151649,
+  "<|box_start|>": 151648,
+  "<|endoftext|>": 151643,
+  "<|file_sep|>": 151664,
+  "<|fim_middle|>": 151660,
+  "<|fim_pad|>": 151662,
+  "<|fim_prefix|>": 151659,
+  "<|fim_suffix|>": 151661,
+  "<|im_end|>": 151645,
+  "<|im_start|>": 151644,
+  "<|image_pad|>": 151655,
+  "<|object_ref_end|>": 151647,
+  "<|object_ref_start|>": 151646,
+  "<|quad_end|>": 151651,
+  "<|quad_start|>": 151650,
+  "<|repo_name|>": 151663,
+  "<|video_pad|>": 151656,
+  "<|vision_end|>": 151653,
+  "<|vision_pad|>": 151654,
+  "<|vision_start|>": 151652
+}

lora/lora-stage2/base_model.txt ADDED Viewed

	@@ -0,0 +1 @@


1	+ /data/haobin/pky_train/qwen3/Qwen3-ASR-1.7B

lora/lora-stage2/chat_template.jinja ADDED Viewed

	@@ -0,0 +1,31 @@

+{%- set ns = namespace(system_text="") -%}
+{%- for m in messages -%}
+  {%- if m.role == 'system' -%}
+    {%- if m.content is string -%}
+      {%- set ns.system_text = ns.system_text + m.content -%}
+    {%- else -%}
+      {%- for c in m.content -%}
+        {%- if c.type == 'text' and (c.text is defined) -%}
+          {%- set ns.system_text = ns.system_text + c.text -%}
+        {%- endif -%}
+      {%- endfor -%}
+    {%- endif -%}
+  {%- endif -%}
+{%- endfor -%}
+{%- set ns2 = namespace(audio_tokens="") -%}
+{%- for m in messages -%}
+  {%- if m.content is not string -%}
+    {%- for c in m.content -%}
+      {%- if c.type == 'audio' or ('audio' in c) or ('audio_url' in c) -%}
+        {%- set ns2.audio_tokens = ns2.audio_tokens + "<|audio_start|><|audio_pad|><|audio_end|>" -%}
+      {%- endif -%}
+    {%- endfor -%}
+  {%- endif -%}
+{%- endfor -%}
+{{- '<|im_start|>system\n' + (ns.system_text if ns.system_text is string else '') + '<|im_end|>\n' -}}
+{{- '<|im_start|>user\n' + ns2.audio_tokens + '<|im_end|>\n' -}}
+{%- if add_generation_prompt -%}
+{{- '<|im_start|>assistant\n' -}}
+{%- endif -%}

lora/lora-stage2/chat_template.json ADDED Viewed

	@@ -0,0 +1 @@

lora/lora-stage2/config.json ADDED Viewed

	@@ -0,0 +1,221 @@

+{
+    "architectures": [
+      "Qwen3ASRForConditionalGeneration"
+    ],
+    "model_type": "qwen3_asr",
+    "support_languages": [
+      "Chinese",
+      "English",
+      "Cantonese",
+      "Arabic",
+      "German",
+      "French",
+      "Spanish",
+      "Portuguese",
+      "Indonesian",
+      "Italian",
+      "Korean",
+      "Russian",
+      "Thai",
+      "Vietnamese",
+      "Japanese",
+      "Turkish",
+      "Hindi",
+      "Malay",
+      "Dutch",
+      "Swedish",
+      "Danish",
+      "Finnish",
+      "Polish",
+      "Czech",
+      "Filipino",
+      "Persian",
+      "Greek",
+      "Romanian",
+      "Hungarian",
+      "Macedonian"
+    ],
+    "thinker_config": {
+      "model_type": "qwen3_asr",
+      "architectures": [
+        "Qwen3ASRForConditionalGeneration"
+      ],
+      "audio_config": {
+        "_name_or_path": "",
+        "activation_dropout": 0,
+        "activation_function": "gelu",
+        "add_cross_attention": false,
+        "architectures": null,
+        "attention_dropout": 0,
+        "bad_words_ids": null,
+        "begin_suppress_tokens": null,
+        "bos_token_id": null,
+        "chunk_size_feed_forward": 0,
+        "conv_chunksize": 500,
+        "cross_attention_hidden_size": null,
+        "d_model": 1024,
+        "decoder_start_token_id": null,
+        "diversity_penalty": 0.0,
+        "do_sample": false,
+        "downsample_hidden_size": 480,
+        "dropout": 0,
+        "dtype": null,
+        "early_stopping": false,
+        "encoder_attention_heads": 16,
+        "encoder_ffn_dim": 4096,
+        "encoder_layers": 24,
+        "encoder_no_repeat_ngram_size": 0,
+        "eos_token_id": null,
+        "exponential_decay_length_penalty": null,
+        "finetuning_task": null,
+        "forced_bos_token_id": null,
+        "forced_eos_token_id": null,
+        "id2label": {
+          "0": "LABEL_0",
+          "1": "LABEL_1"
+        },
+        "initializer_range": 0.02,
+        "is_decoder": false,
+        "is_encoder_decoder": false,
+        "label2id": {
+          "LABEL_0": 0,
+          "LABEL_1": 1
+        },
+        "length_penalty": 1.0,
+        "max_length": 20,
+        "max_source_positions": 1500,
+        "min_length": 0,
+        "model_type": "qwen3_asr_audio_encoder",
+        "n_window": 50,
+        "n_window_infer": 800,
+        "no_repeat_ngram_size": 0,
+        "num_beam_groups": 1,
+        "num_beams": 1,
+        "num_hidden_layers": 24,
+        "num_mel_bins": 128,
+        "num_return_sequences": 1,
+        "output_attentions": false,
+        "output_dim": 2048,
+        "output_hidden_states": false,
+        "output_scores": false,
+        "pad_token_id": null,
+        "prefix": null,
+        "problem_type": null,
+        "pruned_heads": {},
+        "remove_invalid_values": false,
+        "repetition_penalty": 1.0,
+        "return_dict": true,
+        "return_dict_in_generate": false,
+        "scale_embedding": false,
+        "sep_token_id": null,
+        "suppress_tokens": null,
+        "task_specific_params": null,
+        "temperature": 1.0,
+        "tf_legacy_loss": false,
+        "tie_encoder_decoder": false,
+        "tie_word_embeddings": true,
+        "tokenizer_class": null,
+        "top_k": 50,
+        "top_p": 1.0,
+        "torchscript": false,
+        "typical_p": 1.0,
+        "use_bfloat16": false
+      },
+      "audio_end_token_id": 151670,
+      "audio_start_token_id": 151669,
+      "audio_token_id": 151676,
+      "dtype": "bfloat16",
+      "initializer_range": 0.02,
+      "text_config": {
+        "_name_or_path": "",
+        "add_cross_attention": false,
+        "architectures": null,
+        "attention_bias": false,
+        "attention_dropout": 0.0,
+        "bad_words_ids": null,
+        "begin_suppress_tokens": null,
+        "bos_token_id": null,
+        "chunk_size_feed_forward": 0,
+        "cross_attention_hidden_size": null,
+        "decoder_start_token_id": null,
+        "diversity_penalty": 0.0,
+        "do_sample": false,
+        "dtype": null,
+        "early_stopping": false,
+        "encoder_no_repeat_ngram_size": 0,
+        "eos_token_id": null,
+        "exponential_decay_length_penalty": null,
+        "finetuning_task": null,
+        "forced_bos_token_id": null,
+        "forced_eos_token_id": null,
+        "head_dim": 128,
+        "hidden_act": "silu",
+        "hidden_size": 2048,
+        "id2label": {
+          "0": "LABEL_0",
+          "1": "LABEL_1"
+        },
+        "initializer_range": 0.02,
+        "intermediate_size": 6144,
+        "is_decoder": false,
+        "is_encoder_decoder": false,
+        "label2id": {
+          "LABEL_0": 0,
+          "LABEL_1": 1
+        },
+        "length_penalty": 1.0,
+        "max_length": 20,
+        "max_position_embeddings": 65536,
+        "min_length": 0,
+        "model_type": "qwen3",
+        "no_repeat_ngram_size": 0,
+        "num_attention_heads": 16,
+        "num_beam_groups": 1,
+        "num_beams": 1,
+        "num_hidden_layers": 28,
+        "num_key_value_heads": 8,
+        "num_return_sequences": 1,
+        "output_attentions": false,
+        "output_hidden_states": false,
+        "output_scores": false,
+        "pad_token_id": null,
+        "prefix": null,
+        "problem_type": null,
+        "pruned_heads": {},
+        "remove_invalid_values": false,
+        "repetition_penalty": 1.0,
+        "return_dict": true,
+        "return_dict_in_generate": false,
+        "rms_norm_eps": 1e-06,
+        "rope_scaling": {
+          "interleaved": true,
+          "mrope_interleaved": true,
+          "mrope_section": [
+            24,
+            20,
+            20
+          ],
+          "rope_type": "default",
+          "type": "default"
+        },
+        "rope_theta": 1000000,
+        "sep_token_id": null,
+        "suppress_tokens": null,
+        "task_specific_params": null,
+        "temperature": 1.0,
+        "tf_legacy_loss": false,
+        "tie_encoder_decoder": false,
+        "tie_word_embeddings": true,
+        "tokenizer_class": null,
+        "top_k": 50,
+        "top_p": 1.0,
+        "torchscript": false,
+        "typical_p": 1.0,
+        "use_bfloat16": false,
+        "use_cache": true,
+        "vocab_size": 151936
+      }
+    },
+    "transformers_version": "4.57.6"
+  }

lora/lora-stage2/generation_config.json ADDED Viewed

	@@ -0,0 +1,7 @@

+{
+  "_from_model_config": true,
+  "eos_token_id": [151643,151645],
+  "pad_token_id": 151643,
+  "do_sample": false,
+  "temperature": 0.000001
+}

lora/lora-stage2/merged_from_lora.txt ADDED Viewed

	@@ -0,0 +1 @@


1	+ /data/haobin/pky_train/qwen3/out_qwen3-asr-lora-0317_550000_wer3_towerb4+proj_2gpu_bs128/checkpoint-1000

lora/lora-stage2/merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

lora/lora-stage2/optimizer.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:b26cb33d8c7aefdee4dcd88af58551b5a01e16c9f852a1c1ffb0d1a47e6421b4
+size 83695117

lora/lora-stage2/preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,14 @@

+{
+  "chunk_length": 30,
+  "dither": 0.0,
+  "feature_extractor_type": "WhisperFeatureExtractor",
+  "feature_size": 128,
+  "hop_length": 160,
+  "n_fft": 400,
+  "n_samples": 480000,
+  "nb_max_frames": 3000,
+  "padding_side": "right",
+  "padding_value": 0.0,
+  "processor_class": "Qwen3ASRProcessor",
+  "return_attention_mask": true
+}

lora/lora-stage2/rng_state_0.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:de015da1ba6a4dc8cf66420b3b9b378bc07585bfb14a0c37fb50e723424b9768
+size 14917

lora/lora-stage2/rng_state_1.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:681f2e7cc7c3d884111a86a3bcdeeaea97b22ebf60e4f765788ee5cbeb94e2d9
+size 14917

lora/lora-stage2/scheduler.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:2a7077d452a1df5790a83102fc7a743c5150e80f24610df63abd069404ebe93a
+size 1465

lora/lora-stage2/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,44 @@

+{
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|object_ref_start|>",
+    "<|object_ref_end|>",
+    "<|box_start|>",
+    "<|box_end|>",
+    "<|quad_start|>",
+    "<|quad_end|>",
+    "<|vision_start|>",
+    "<|vision_end|>",
+    "<|vision_pad|>",
+    "<|image_pad|>",
+    "<|video_pad|>",
+    "<|audio_start|>",
+    "<|audio_end|>",
+    "<tts_pad>",
+    "<tts_text_bos>",
+    "<tts_text_bos_single>",
+    "<|audio_pad|>"
+  ],
+  "audio_bos_token": "<|audio_start|>",
+  "audio_eos_token": "<|audio_end|>",
+  "audio_token": "<|audio_pad|>",
+  "eos_token": {
+    "content": "<|im_end|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "image_token": "<|image_pad|>",
+  "pad_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "video_token": "<|video_pad|>",
+  "vision_bos_token": "<|vision_start|>",
+  "vision_eos_token": "<|vision_end|>"
+}

lora/lora-stage2/tokenizer.json ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:0499602714160467f2d68b910651d6216020689f1e016be87a2d0019ee3baeab
+size 11429499

lora/lora-stage2/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,549 @@

+{
+  "add_bos_token": false,
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "151643": {
+      "content": "<|endoftext|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151644": {
+      "content": "<|im_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151645": {
+      "content": "<|im_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151646": {
+      "content": "<|object_ref_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151647": {
+      "content": "<|object_ref_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151648": {
+      "content": "<|box_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151649": {
+      "content": "<|box_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151650": {
+      "content": "<|quad_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151651": {
+      "content": "<|quad_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151652": {
+      "content": "<|vision_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151653": {
+      "content": "<|vision_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151654": {
+      "content": "<|vision_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151655": {
+      "content": "<|image_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151656": {
+      "content": "<|video_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151657": {
+      "content": "<tool_call>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151658": {
+      "content": "</tool_call>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151659": {
+      "content": "<|fim_prefix|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151660": {
+      "content": "<|fim_middle|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151661": {
+      "content": "<|fim_suffix|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151662": {
+      "content": "<|fim_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151663": {
+      "content": "<|repo_name|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151664": {
+      "content": "<|file_sep|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151665": {
+      "content": "<tool_response>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151666": {
+      "content": "</tool_response>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151667": {
+      "content": "<think>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151668": {
+      "content": "</think>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151669": {
+      "content": "<|audio_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151670": {
+      "content": "<|audio_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151671": {
+      "content": "<tts_pad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151672": {
+      "content": "<tts_text_bos>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151673": {
+      "content": "<tts_text_eod>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151674": {
+      "content": "<tts_text_bos_single>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151675": {
+      "content": "<non_speech>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151676": {
+      "content": "<|audio_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151677": {
+      "content": "<blank1>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151678": {
+      "content": "<blank2>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151679": {
+      "content": "<blank3>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151680": {
+      "content": "<blank4>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151681": {
+      "content": "<blank5>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151682": {
+      "content": "<blank6>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151683": {
+      "content": "<blank7>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151684": {
+      "content": "<blank8>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151685": {
+      "content": "<blank9>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151686": {
+      "content": "<blank10>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151687": {
+      "content": "<blank11>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151688": {
+      "content": "<blank12>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151689": {
+      "content": "<blank13>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151690": {
+      "content": "<blank14>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151691": {
+      "content": "<blank15>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151692": {
+      "content": "<blank16>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151693": {
+      "content": "<blank17>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151694": {
+      "content": "<blank18>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151695": {
+      "content": "<blank19>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151696": {
+      "content": "<blank20>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151697": {
+      "content": "<blank21>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151698": {
+      "content": "<blank22>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151699": {
+      "content": "<blank23>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151700": {
+      "content": "<blank24>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151701": {
+      "content": "<blank25>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151702": {
+      "content": "<blank26>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151703": {
+      "content": "<blank27>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151704": {
+      "content": "<asr_text>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    }
+  },
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|object_ref_start|>",
+    "<|object_ref_end|>",
+    "<|box_start|>",
+    "<|box_end|>",
+    "<|quad_start|>",
+    "<|quad_end|>",
+    "<|vision_start|>",
+    "<|vision_end|>",
+    "<|vision_pad|>",
+    "<|image_pad|>",
+    "<|video_pad|>",
+    "<|audio_start|>",
+    "<|audio_end|>",
+    "<tts_pad>",
+    "<tts_text_bos>",
+    "<tts_text_bos_single>",
+    "<|audio_pad|>"
+  ],
+  "audio_bos_token": "<|audio_start|>",
+  "audio_eos_token": "<|audio_end|>",
+  "audio_token": "<|audio_pad|>",
+  "bos_token": null,
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<|im_end|>",
+  "errors": "replace",
+  "extra_special_tokens": {
+    "audio_bos_token": "<|audio_start|>",
+    "audio_eos_token": "<|audio_end|>",
+    "audio_token": "<|audio_pad|>",
+    "image_token": "<|image_pad|>",
+    "video_token": "<|video_pad|>",
+    "vision_bos_token": "<|vision_start|>",
+    "vision_eos_token": "<|vision_end|>"
+  },
+  "image_token": "<|image_pad|>",
+  "model_max_length": 131072,
+  "pad_token": "<|endoftext|>",
+  "processor_class": "Qwen3ASRProcessor",
+  "split_special_tokens": false,
+  "tokenizer_class": "Qwen2Tokenizer",
+  "unk_token": null,
+  "video_token": "<|video_pad|>",
+  "vision_bos_token": "<|vision_start|>",
+  "vision_eos_token": "<|vision_end|>"
+}

lora/lora-stage2/trainer_state.json ADDED Viewed

The diff for this file is too large to render. See raw diff

lora/lora-stage2/vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff

lora/lora-stage3/README.md ADDED Viewed

	@@ -0,0 +1,207 @@

+---
+base_model: /data/haobin/Qwen3-ASR/Qwen3-ASR-1.7B-lora-merged
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:/data/haobin/Qwen3-ASR/Qwen3-ASR-1.7B-lora-merged
+- lora
+- transformers
+---
+# Model Card for Model ID
+<!-- Provide a quick summary of what the model is/does. -->
+## Model Details
+### Model Description
+<!-- Provide a longer summary of what this model is. -->
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+### Model Sources [optional]
+<!-- Provide the basic links for the model. -->
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+## Uses
+<!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
+### Direct Use
+<!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
+[More Information Needed]
+### Downstream Use [optional]
+<!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
+[More Information Needed]
+### Out-of-Scope Use
+<!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
+[More Information Needed]
+## Bias, Risks, and Limitations
+<!-- This section is meant to convey both technical and sociotechnical limitations. -->
+[More Information Needed]
+### Recommendations
+<!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+## How to Get Started with the Model
+Use the code below to get started with the model.
+[More Information Needed]
+## Training Details
+### Training Data
+<!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
+[More Information Needed]
+### Training Procedure
+<!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
+#### Preprocessing [optional]
+[More Information Needed]
+#### Training Hyperparameters
+- **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
+#### Speeds, Sizes, Times [optional]
+<!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
+[More Information Needed]
+## Evaluation
+<!-- This section describes the evaluation protocols and provides the results. -->
+### Testing Data, Factors & Metrics
+#### Testing Data
+<!-- This should link to a Dataset Card if possible. -->
+[More Information Needed]
+#### Factors
+<!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
+[More Information Needed]
+#### Metrics
+<!-- These are the evaluation metrics being used, ideally with a description of why. -->
+[More Information Needed]
+### Results
+[More Information Needed]
+#### Summary
+## Model Examination [optional]
+<!-- Relevant interpretability work for the model goes here -->
+[More Information Needed]
+## Environmental Impact
+<!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+## Technical Specifications [optional]
+### Model Architecture and Objective
+[More Information Needed]
+### Compute Infrastructure
+[More Information Needed]
+#### Hardware
+[More Information Needed]
+#### Software
+[More Information Needed]
+## Citation [optional]
+<!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
+**BibTeX:**
+[More Information Needed]
+**APA:**
+[More Information Needed]
+## Glossary [optional]
+<!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
+[More Information Needed]
+## More Information [optional]
+[More Information Needed]
+## Model Card Authors [optional]
+[More Information Needed]
+## Model Card Contact
+[More Information Needed]
+### Framework versions
+- PEFT 0.18.1

lora/lora-stage3/adapter_config.json ADDED Viewed

	@@ -0,0 +1,38 @@

+{
+  "alora_invocation_tokens": null,
+  "alpha_pattern": {},
+  "arrow_config": null,
+  "auto_mapping": null,
+  "base_model_name_or_path": "/data/haobin/Qwen3-ASR/Qwen3-ASR-1.7B-lora-merged",
+  "bias": "none",
+  "corda_config": null,
+  "ensure_weight_tying": false,
+  "eva_config": null,
+  "exclude_modules": null,
+  "fan_in_fan_out": false,
+  "inference_mode": true,
+  "init_lora_weights": true,
+  "layer_replication": null,
+  "layers_pattern": null,
+  "layers_to_transform": null,
+  "loftq_config": {},
+  "lora_alpha": 32,
+  "lora_bias": false,
+  "lora_dropout": 0.05,
+  "megatron_config": null,
+  "megatron_core": "megatron.core",
+  "modules_to_save": [],
+  "peft_type": "LORA",
+  "peft_version": "0.18.1",
+  "qalora_group_size": 16,
+  "r": 8,
+  "rank_pattern": {},
+  "revision": null,
+  "target_modules": "^(thinker\\.model(?=\\.).*\\.(k_proj|q_proj|o_proj|up_proj|down_proj|v_proj|gate_proj)|(?!(thinker.audio_tower.proj1|thinker.audio_tower.proj2))thinker\\.audio_tower(?=\\.).*\\.(fc1|out_proj|proj1|k_proj|q_proj|fc2|proj2|v_proj|conv_out)|thinker\\.audio_tower\\.proj1(?=\\.)|thinker\\.audio_tower\\.proj2(?=\\.))$",
+  "target_parameters": null,
+  "task_type": "CAUSAL_LM",
+  "trainable_token_indices": null,
+  "use_dora": false,
+  "use_qalora": false,
+  "use_rslora": false
+}

lora/lora-stage3/adapter_model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:1b5507acb5bb51851c4db58504cac3dcc748dbc37210b986e93624bb9ea115b0
+size 49395592

lora/lora-stage3/additional_config.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"lora_dtype": null, "lorap_lr_ratio": null, "lorap_emb_lr": 1e-06}

lora/lora-stage3/args.json ADDED Viewed

	@@ -0,0 +1,502 @@

+{
+  "output_dir": "/data/haobin/pky_train/qwen3_swift/pky_out/qwen3asr_dapo_reward5_3x8x8_12gen_3GPU/v3-20260410-173721",
+  "overwrite_output_dir": false,
+  "do_train": false,
+  "do_eval": false,
+  "do_predict": false,
+  "eval_strategy": "steps",
+  "prediction_loss_only": false,
+  "per_device_train_batch_size": 4,
+  "per_device_eval_batch_size": 4,
+  "per_gpu_train_batch_size": null,
+  "per_gpu_eval_batch_size": null,
+  "gradient_accumulation_steps": 16,
+  "eval_accumulation_steps": null,
+  "eval_delay": 0,
+  "torch_empty_cache_steps": null,
+  "learning_rate": 5e-05,
+  "weight_decay": 0.1,
+  "adam_beta1": 0.9,
+  "adam_beta2": 0.95,
+  "adam_epsilon": 1e-08,
+  "max_grad_norm": 1.0,
+  "num_train_epochs": 3.0,
+  "max_steps": -1,
+  "lr_scheduler_type": "cosine",
+  "lr_scheduler_kwargs": null,
+  "warmup_ratio": 0.03,
+  "warmup_steps": 0,
+  "log_level": "passive",
+  "log_level_replica": "warning",
+  "log_on_each_node": true,
+  "logging_dir": "/data/haobin/pky_train/qwen3_swift/pky_out/qwen3asr_dapo_reward5_3x8x8_12gen_3GPU/v3-20260410-173721/runs",
+  "logging_strategy": "steps",
+  "logging_first_step": true,
+  "logging_steps": 5,
+  "logging_nan_inf_filter": true,
+  "save_strategy": "steps",
+  "save_steps": 20.0,
+  "save_total_limit": null,
+  "save_safetensors": true,
+  "save_on_each_node": false,
+  "save_only_model": false,
+  "restore_callback_states_from_checkpoint": false,
+  "no_cuda": false,
+  "use_cpu": false,
+  "use_mps_device": false,
+  "seed": 42,
+  "data_seed": 42,
+  "jit_mode_eval": false,
+  "bf16": true,
+  "fp16": false,
+  "fp16_opt_level": "O1",
+  "half_precision_backend": "auto",
+  "bf16_full_eval": false,
+  "fp16_full_eval": false,
+  "tf32": null,
+  "local_rank": 0,
+  "ddp_backend": null,
+  "tpu_num_cores": null,
+  "tpu_metrics_debug": false,
+  "debug": null,
+  "dataloader_drop_last": false,
+  "eval_steps": 20.0,
+  "dataloader_num_workers": null,
+  "dataloader_prefetch_factor": null,
+  "past_index": -1,
+  "run_name": "qwen3asr_dapo_reward5_3x8x8_12gen_3GPU",
+  "disable_tqdm": null,
+  "remove_unused_columns": false,
+  "label_names": null,
+  "load_best_model_at_end": false,
+  "metric_for_best_model": "loss",
+  "greater_is_better": false,
+  "ignore_data_skip": false,
+  "fsdp": [],
+  "fsdp_min_num_params": 0,
+  "fsdp_config": null,
+  "fsdp_transformer_layer_cls_to_wrap": null,
+  "accelerator_config": {
+    "dispatch_batches": false
+  },
+  "parallelism_config": null,
+  "deepspeed": null,
+  "label_smoothing_factor": 0.0,
+  "optim": "adamw_torch_fused",
+  "optim_args": null,
+  "adafactor": false,
+  "group_by_length": false,
+  "length_column_name": "length",
+  "report_to": [
+    "wandb"
+  ],
+  "project": "huggingface",
+  "trackio_space_id": "trackio",
+  "ddp_find_unused_parameters": null,
+  "ddp_bucket_cap_mb": null,
+  "ddp_broadcast_buffers": null,
+  "dataloader_pin_memory": true,
+  "dataloader_persistent_workers": false,
+  "skip_memory_metrics": true,
+  "use_legacy_prediction_loop": false,
+  "push_to_hub": false,
+  "resume_from_checkpoint": null,
+  "hub_model_id": null,
+  "hub_strategy": "every_save",
+  "hub_token": null,
+  "hub_private_repo": null,
+  "hub_always_push": false,
+  "hub_revision": null,
+  "gradient_checkpointing": true,
+  "gradient_checkpointing_kwargs": null,
+  "include_inputs_for_metrics": false,
+  "include_for_metrics": [],
+  "eval_do_concat_batches": true,
+  "fp16_backend": "auto",
+  "push_to_hub_model_id": null,
+  "push_to_hub_organization": null,
+  "push_to_hub_token": null,
+  "mp_parameters": "",
+  "auto_find_batch_size": false,
+  "full_determinism": false,
+  "torchdynamo": null,
+  "ray_scope": "last",
+  "ddp_timeout": 18000000,
+  "torch_compile": false,
+  "torch_compile_backend": null,
+  "torch_compile_mode": null,
+  "include_tokens_per_second": false,
+  "include_num_input_tokens_seen": false,
+  "neftune_noise_alpha": null,
+  "optim_target_modules": null,
+  "batch_eval_metrics": false,
+  "eval_on_start": false,
+  "use_liger_kernel": false,
+  "liger_kernel_config": null,
+  "eval_use_gather_object": false,
+  "average_tokens_across_devices": true,
+  "sortish_sampler": false,
+  "predict_with_generate": false,
+  "generation_max_length": null,
+  "generation_num_beams": null,
+  "generation_config": null,
+  "tuner_backend": "peft",
+  "vit_gradient_checkpointing": null,
+  "router_aux_loss_coef": 0.0,
+  "enable_dft_loss": false,
+  "enable_channel_loss": false,
+  "safe_serialization": true,
+  "max_shard_size": "5GB",
+  "check_model": true,
+  "acc_strategy": "token",
+  "train_dataloader_shuffle": true,
+  "max_epochs": null,
+  "aligner_lr": null,
+  "vit_lr": null,
+  "use_logits_to_keep": null,
+  "ds3_gather_for_generation": true,
+  "resume_only_model": false,
+  "optimizer": null,
+  "loss_type": "dapo",
+  "eval_metric": null,
+  "callbacks": [],
+  "early_stop_interval": null,
+  "eval_use_evalscope": false,
+  "eval_dataset": [],
+  "eval_dataset_args": null,
+  "eval_limit": null,
+  "eval_generation_config": null,
+  "extra_eval_args": null,
+  "tuner_type": "lora",
+  "use_galore": false,
+  "galore_target_modules": null,
+  "galore_rank": 128,
+  "galore_update_proj_gap": 50,
+  "galore_scale": 1.0,
+  "galore_proj_type": "std",
+  "galore_optim_per_parameter": false,
+  "galore_with_embedding": false,
+  "galore_quantization": false,
+  "galore_proj_quant": false,
+  "galore_proj_bits": 4,
+  "galore_proj_group_size": 256,
+  "galore_cos_threshold": 0.4,
+  "galore_gamma_proj": 2,
+  "galore_queue_size": 5,
+  "lisa_activated_layers": 0,
+  "lisa_step_interval": 20,
+  "use_flash_ckpt": false,
+  "use_ray": false,
+  "ray_exp_name": null,
+  "device_groups": null,
+  "model": "/data/haobin/Qwen3-ASR/Qwen3-ASR-1.7B-lora-merged",
+  "model_type": "my_qwen3_asr_rl",
+  "model_revision": null,
+  "task_type": "causal_lm",
+  "torch_dtype": "bfloat16",
+  "attn_impl": null,
+  "experts_impl": null,
+  "new_special_tokens": [],
+  "num_labels": null,
+  "problem_type": null,
+  "rope_scaling": null,
+  "device_map": null,
+  "max_memory": {},
+  "max_model_len": null,
+  "local_repo_path": null,
+  "init_strategy": null,
+  "template": "my_qwen3_asr_rl",
+  "system": null,
+  "max_length": 65536,
+  "truncation_strategy": "delete",
+  "max_pixels": null,
+  "agent_template": null,
+  "norm_bbox": null,
+  "use_chat_template": true,
+  "padding_side": "left",
+  "padding_free": false,
+  "loss_scale": "last_round",
+  "sequence_parallel_size": 1,
+  "template_backend": "swift",
+  "response_prefix": null,
+  "enable_thinking": null,
+  "add_non_thinking_prefix": true,
+  "dataset": [
+    "/data/haobin/batch_process/lora_0323_10w+55w+error+syn_with_domain_train90_targeted_rl_train90_loramerged_basewer_271.jsonl"
+  ],
+  "val_dataset": [
+    "/data/haobin/batch_process/lora_0323_10w+55w+error+syn_with_domain_train90_targeted_rl_val5_sample5p.jsonl"
+  ],
+  "cached_dataset": [],
+  "cached_val_dataset": [],
+  "split_dataset_ratio": 0.0,
+  "dataset_num_proc": 1,
+  "load_from_cache_file": false,
+  "dataset_shuffle": true,
+  "val_dataset_shuffle": false,
+  "streaming": false,
+  "interleave_prob": null,
+  "stopping_strategy": "first_exhausted",
+  "shuffle_buffer_size": 1000,
+  "download_mode": "reuse_dataset_if_exists",
+  "columns": {},
+  "strict": false,
+  "model_name": null,
+  "model_author": null,
+  "custom_dataset_info": [],
+  "quant_method": null,
+  "quant_bits": null,
+  "hqq_axis": null,
+  "bnb_4bit_compute_dtype": "bfloat16",
+  "bnb_4bit_quant_type": "nf4",
+  "bnb_4bit_use_double_quant": true,
+  "bnb_4bit_quant_storage": null,
+  "max_new_tokens": 256,
+  "temperature": 0.5,
+  "top_k": 50,
+  "top_p": 0.95,
+  "repetition_penalty": 1.08,
+  "num_beams": 1,
+  "stream": false,
+  "stop_words": [],
+  "logprobs": false,
+  "top_logprobs": null,
+  "structured_outputs_regex": null,
+  "train_type": "lora",
+  "adapters": [],
+  "external_plugins": [
+    "/data/haobin/pky_train/qwen3_swift/my_qwen3_asr_dapo_register.py",
+    "/data/haobin/pky_train/qwen3_swift/qwen3_RL_reward5.py"
+  ],
+  "custom_register_path": [],
+  "model_kwargs": {},
+  "load_args": false,
+  "load_data_args": false,
+  "packing": false,
+  "packing_length": null,
+  "packing_num_proc": 1,
+  "lazy_tokenize": true,
+  "use_hf": false,
+  "ignore_args_error": false,
+  "use_swift_lora": false,
+  "freeze_parameters": [],
+  "freeze_parameters_regex": null,
+  "freeze_parameters_ratio": 0.0,
+  "trainable_parameters": [],
+  "trainable_parameters_regex": null,
+  "freeze_llm": false,
+  "freeze_vit": false,
+  "freeze_aligner": false,
+  "target_modules": [
+    "all-linear"
+  ],
+  "target_regex": null,
+  "target_parameters": null,
+  "modules_to_save": [],
+  "lora_rank": 8,
+  "lora_alpha": 32,
+  "lora_dropout": 0.05,
+  "lora_bias": "none",
+  "lora_dtype": null,
+  "lorap_lr_ratio": null,
+  "use_rslora": false,
+  "use_dora": false,
+  "lora_ga_batch_size": 2,
+  "lora_ga_iters": 2,
+  "lora_ga_max_length": 1024,
+  "lora_ga_direction": "ArB2r",
+  "lora_ga_scale": "stable",
+  "lora_ga_stable_gamma": 16,
+  "init_weights": true,
+  "fourier_n_frequency": 2000,
+  "fourier_scaling": 300.0,
+  "boft_block_size": 4,
+  "boft_block_num": 0,
+  "boft_n_butterfly_factor": 1,
+  "boft_dropout": 0.0,
+  "vera_rank": 256,
+  "vera_projection_prng_key": 0,
+  "vera_dropout": 0.0,
+  "vera_d_initial": 0.1,
+  "adapter_act": "gelu",
+  "adapter_length": 128,
+  "adalora_target_r": 8,
+  "adalora_init_r": 12,
+  "adalora_tinit": 0,
+  "adalora_tfinal": 0,
+  "adalora_deltaT": 1,
+  "adalora_beta1": 0.85,
+  "adalora_beta2": 0.85,
+  "adalora_orth_reg_weight": 0.5,
+  "llamapro_num_new_blocks": 4,
+  "llamapro_num_groups": null,
+  "reft_layer_key": null,
+  "reft_layers": null,
+  "reft_rank": 4,
+  "reft_intervention_type": "LoreftIntervention",
+  "reft_args": null,
+  "swanlab_token": null,
+  "swanlab_project": "ms-swift",
+  "swanlab_workspace": null,
+  "swanlab_exp_name": null,
+  "swanlab_notification_method": null,
+  "swanlab_webhook_url": null,
+  "swanlab_secret": null,
+  "swanlab_sender_email": null,
+  "swanlab_receiver_email": null,
+  "swanlab_smtp_server": null,
+  "swanlab_smtp_port": null,
+  "swanlab_email_language": "zh",
+  "swanlab_mode": "cloud",
+  "add_version": true,
+  "create_checkpoint_symlink": false,
+  "zero_hpz_partition_size": null,
+  "deepspeed_autotp_size": null,
+  "reward_model": null,
+  "reward_adapters": [],
+  "reward_model_type": null,
+  "reward_model_revision": null,
+  "num_ppo_epochs": 4,
+  "whiten_rewards": false,
+  "kl_coef": 0.05,
+  "cliprange": 0.2,
+  "vf_coef": 0.1,
+  "cliprange_value": 0.2,
+  "gamma": 1.0,
+  "lam": 0.95,
+  "num_mini_batches": 1,
+  "local_rollout_forward_batch_size": 64,
+  "num_sample_generations": 10,
+  "response_length": 256,
+  "missing_eos_penalty": null,
+  "vllm_gpu_memory_utilization": 0.9,
+  "vllm_tensor_parallel_size": 1,
+  "vllm_pipeline_parallel_size": 1,
+  "vllm_enable_expert_parallel": false,
+  "vllm_max_num_seqs": null,
+  "vllm_max_model_len": null,
+  "vllm_disable_custom_all_reduce": true,
+  "vllm_enforce_eager": false,
+  "vllm_limit_mm_per_prompt": null,
+  "vllm_max_lora_rank": 16,
+  "vllm_enable_prefix_caching": true,
+  "vllm_use_async_engine": null,
+  "vllm_quantization": null,
+  "vllm_reasoning_parser": null,
+  "vllm_disable_cascade_attn": false,
+  "vllm_mm_processor_cache_gb": null,
+  "vllm_speculative_config": null,
+  "vllm_engine_kwargs": {},
+  "vllm_data_parallel_size": 1,
+  "use_vllm": false,
+  "vllm_mode": null,
+  "vllm_enable_lora": false,
+  "vllm_server_base_url": null,
+  "vllm_server_host": null,
+  "vllm_server_port": [
+    8000
+  ],
+  "vllm_server_timeout": 240.0,
+  "vllm_server_group_port": null,
+  "enable_flattened_weight_sync": true,
+  "async_generate": false,
+  "sleep_level": 0,
+  "move_model_batches": null,
+  "offload_optimizer": false,
+  "offload_model": false,
+  "wandb_log_unique_prompts": null,
+  "epsilon": 0.2,
+  "epsilon_high": 0.28,
+  "delta": null,
+  "cosine_min_len_value_wrong": -0.5,
+  "cosine_max_len_value_wrong": 0.0,
+  "cosine_min_len_value_correct": 1.0,
+  "cosine_max_len_value_correct": 0.5,
+  "cosine_max_len": null,
+  "repetition_n_grams": 3,
+  "repetition_max_penalty": -1.0,
+  "reward_model_plugin": null,
+  "chord_sft_dataset": [],
+  "chord_sft_per_device_train_batch_size": null,
+  "chord_enable_phi_function": false,
+  "chord_mu_warmup_steps": null,
+  "chord_mu_decay_steps": null,
+  "chord_mu_peak": null,
+  "chord_mu_valley": null,
+  "sync_ref_model": false,
+  "ref_model_sync_steps": 512,
+  "ref_model_mixup_alpha": 0.6,
+  "multi_turn_scheduler": null,
+  "max_turns": null,
+  "completion_length_limit_scope": "per_round",
+  "vllm_server_pass_dataset": false,
+  "dynamic_sample": true,
+  "max_resample_times": 4,
+  "overlong_filter": true,
+  "soft_max_length": null,
+  "soft_cache_length": null,
+  "scale_rewards": "group",
+  "log_entropy": false,
+  "top_entropy_quantile": 1.0,
+  "importance_sampling_level": "token",
+  "tau_pos": 1.0,
+  "tau_neg": 1.05,
+  "advantage_estimator": "grpo",
+  "kl_in_reward": false,
+  "generation_batch_size": 48,
+  "steps_per_generation": null,
+  "num_generations_eval": 4,
+  "rollout_importance_sampling_mode": null,
+  "rollout_importance_sampling_threshold": 2.0,
+  "log_rollout_offpolicy_metrics": false,
+  "off_policy_sequence_mask_delta": null,
+  "num_generations": 12,
+  "reward_funcs": [
+    "asr_wer_hallu_len_v5"
+  ],
+  "reward_weights": null,
+  "log_completions": true,
+  "num_iterations": 2,
+  "teacher_model": null,
+  "teacher_adapters": [],
+  "teacher_model_type": null,
+  "teacher_model_revision": null,
+  "teacher_deepspeed": null,
+  "teacher_model_server": null,
+  "rlhf_type": "grpo",
+  "ref_model": null,
+  "ref_adapters": [],
+  "ref_model_type": null,
+  "ref_model_revision": null,
+  "beta": 0.04,
+  "label_smoothing": 0,
+  "max_completion_length": 256,
+  "rpo_alpha": null,
+  "ld_alpha": null,
+  "discopop_tau": 0.05,
+  "loss_weights": null,
+  "cpo_alpha": 1.0,
+  "simpo_gamma": 1,
+  "desirable_weight": 1.0,
+  "undesirable_weight": 1.0,
+  "center_rewards_coefficient": null,
+  "sft_alpha": 0,
+  "lmbda": 0.5,
+  "seq_kd": false,
+  "gkd_logits_topk": null,
+  "offload_teacher_model": false,
+  "swift_version": "4.0.3",
+  "ckpt_dir": null,
+  "rank": 0,
+  "global_world_size": 3,
+  "local_world_size": 3,
+  "model_suffix": "Qwen3-ASR-1.7B-lora-merged",
+  "model_info": "ModelInfo(model_type='my_qwen3_asr_rl', model_dir='/data/haobin/Qwen3-ASR/Qwen3-ASR-1.7B-lora-merged', torch_dtype=torch.bfloat16, max_model_len=65536, quant_method=None, quant_bits=None, rope_scaling={'interleaved': True, 'mrope_interleaved': True, 'mrope_section': [24, 20, 20], 'rope_type': 'default', 'type': 'default'}, is_moe_model=False, is_multimodal=True, config=None, task_type='causal_lm', num_labels=None)",
+  "model_meta": "ModelMeta(model_type='my_qwen3_asr_rl', model_groups=[ModelGroup(models=[Model(ms_model_id='Qwen/Qwen3-ASR-0.6B', hf_model_id=None, model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen3-ASR-1.7B', hf_model_id=None, model_path=None, ms_revision=None, hf_revision=None)], template=None, ignore_patterns=None, requires=None, tags=[])], loader=<class 'my_qwen3_asr_dapo_register.Qwen3ASRRLLoader'>, template='my_qwen3_asr_rl', model_arch=MultiModelKeys(arch_name='my_qwen3_asr_rl', embedding=None, module_list=None, lm_head=None, q_proj=None, k_proj=None, v_proj=None, o_proj=None, attention=None, mlp=None, down_proj=None, qkv_proj=None, qk_proj=None, qa_proj=None, qb_proj=None, kv_proj=None, kva_proj=None, kvb_proj=None, language_model=['thinker.model', 'thinker.lm_head'], aligner=['thinker.audio_tower.proj1', 'thinker.audio_tower.proj2'], vision_tower=['thinker.audio_tower'], generator=[]), architectures=['Qwen3ASRForConditionalGeneration'], additional_saved_files=['generation_config.json', 'preprocessor_config.json', 'processor_config.json', 'tokenizer_config.json', 'tokenizer.json', 'special_tokens_map.json', 'chat_template.json', 'merges.txt', 'vocab.json'], torch_dtype=None, is_multimodal=True, is_reward=False, task_type=None, ignore_patterns=None, requires=['transformers>=4.57', 'qwen-asr', 'librosa'], tags=['audio'])",
+  "model_dir": "/data/haobin/Qwen3-ASR/Qwen3-ASR-1.7B-lora-merged",
+  "template_meta": "TemplateMeta(template_type='my_qwen3_asr_rl', prefix=[], prompt=['{{QUERY}}'], chat_sep=[], suffix=[''], template_cls=<class 'my_qwen3_asr_dapo_register.Qwen3ASRRLTemplate'>, system_prefix=[], default_system=None, auto_add_bos=False, stop_words=[], agent_template='react_en', is_thinking=False, thinking_prefix='', non_thinking_prefix='', history_thinking_prefix='')",
+  "_val_dataset_exists": true,
+  "hub": "<class 'swift.hub.hub.MSHub'>",
+  "evaluation_strategy": "steps",
+  "training_args": "GRPOConfig(output_dir='/data/haobin/pky_train/qwen3_swift/pky_out/qwen3asr_dapo_reward5_3x8x8_12gen_3GPU/v3-20260410-173721', overwrite_output_dir=False, do_train=False, do_eval=True, do_predict=False, eval_strategy=<IntervalStrategy.STEPS: 'steps'>, prediction_loss_only=False, per_device_train_batch_size=4, per_device_eval_batch_size=4, per_gpu_train_batch_size=None, per_gpu_eval_batch_size=None, gradient_accumulation_steps=16, eval_accumulation_steps=None, eval_delay=0, torch_empty_cache_steps=None, learning_rate=5e-05, weight_decay=0.1, adam_beta1=0.9, adam_beta2=0.95, adam_epsilon=1e-08, max_grad_norm=1.0, num_train_epochs=3.0, max_steps=-1, lr_scheduler_type=<SchedulerType.COSINE: 'cosine'>, lr_scheduler_kwargs=None, warmup_ratio=0.03, warmup_steps=0, log_level='passive', log_level_replica='warning', log_on_each_node=True, logging_dir='/data/haobin/pky_train/qwen3_swift/pky_out/qwen3asr_dapo_reward5_3x8x8_12gen_3GPU/v3-20260410-173721/runs', logging_strategy=<IntervalStrategy.STEPS: 'steps'>, logging_first_step=True, logging_steps=5, logging_nan_inf_filter=True, save_strategy=<SaveStrategy.STEPS: 'steps'>, save_steps=20, save_total_limit=None, save_safetensors=True, save_on_each_node=False, save_only_model=False, restore_callback_states_from_checkpoint=False, no_cuda=False, use_cpu=False, use_mps_device=False, seed=42, data_seed=42, jit_mode_eval=False, bf16=True, fp16=False, fp16_opt_level='O1', half_precision_backend='auto', bf16_full_eval=False, fp16_full_eval=False, tf32=None, local_rank=0, ddp_backend=None, tpu_num_cores=None, tpu_metrics_debug=False, debug=[], dataloader_drop_last=True, eval_steps=20, dataloader_num_workers=1, dataloader_prefetch_factor=2, past_index=-1, run_name='qwen3asr_dapo_reward5_3x8x8_12gen_3GPU', disable_tqdm=False, remove_unused_columns=False, label_names=None, load_best_model_at_end=False, metric_for_best_model='loss', greater_is_better=False, ignore_data_skip=False, fsdp=[], fsdp_min_num_params=0, fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False}, fsdp_transformer_layer_cls_to_wrap=None, accelerator_config=AcceleratorConfig(split_batches=False, dispatch_batches=False, even_batches=True, use_seedable_sampler=True, non_blocking=False, gradient_accumulation_kwargs=None, use_configured_state=False), parallelism_config=None, deepspeed=None, label_smoothing_factor=0.0, optim=<OptimizerNames.ADAMW_TORCH_FUSED: 'adamw_torch_fused'>, optim_args=None, adafactor=False, group_by_length=False, length_column_name='length', report_to=['wandb'], project='huggingface', trackio_space_id='trackio', ddp_find_unused_parameters=None, ddp_bucket_cap_mb=None, ddp_broadcast_buffers=None, dataloader_pin_memory=True, dataloader_persistent_workers=False, skip_memory_metrics=True, use_legacy_prediction_loop=False, push_to_hub=False, resume_from_checkpoint=None, hub_model_id=None, hub_strategy=<HubStrategy.EVERY_SAVE: 'every_save'>, hub_token=None, hub_private_repo=None, hub_always_push=False, hub_revision=None, gradient_checkpointing=True, gradient_checkpointing_kwargs=None, include_inputs_for_metrics=False, include_for_metrics=[], eval_do_concat_batches=True, fp16_backend='auto', push_to_hub_model_id=None, push_to_hub_organization=None, push_to_hub_token=None, mp_parameters='', auto_find_batch_size=False, full_determinism=False, torchdynamo=None, ray_scope='last', ddp_timeout=18000000, torch_compile=False, torch_compile_backend=None, torch_compile_mode=None, include_tokens_per_second=None, include_num_input_tokens_seen=None, neftune_noise_alpha=None, optim_target_modules=None, batch_eval_metrics=False, eval_on_start=False, use_liger_kernel=False, liger_kernel_config=None, eval_use_gather_object=False, average_tokens_across_devices=None, model_init_kwargs=None, disable_dropout=False, cast_lm_head_to_fp32=False, num_generations=12, num_generations_eval=4, max_completion_length=256, ds3_gather_for_generation=True, shuffle_dataset=True, generation_batch_size=48, steps_per_generation=4, temperature=0.5, top_p=0.95, top_k=50, min_p=None, generation_kwargs=None, chat_template_kwargs=None, repetition_penalty=1.08, use_transformers_paged=False, cache_implementation=None, use_vllm=False, vllm_mode=None, vllm_model_impl='vllm', vllm_enable_sleep_mode=False, vllm_structured_outputs_regex=None, vllm_server_base_url=None, vllm_server_host=None, vllm_server_port=[8000], vllm_server_timeout=240.0, vllm_group_port=51216, vllm_gpu_memory_utilization=0.9, vllm_max_model_length=None, vllm_tensor_parallel_size=1, beta=0.04, num_iterations=2, epsilon=0.2, delta=None, epsilon_high=0.28, sapo_temperature_neg=1.05, sapo_temperature_pos=1.0, importance_sampling_level='token', reward_weights=None, multi_objective_aggregation='sum_then_normalize', scale_rewards='group', loss_type='dapo', mask_truncated_completions=False, sync_ref_model=False, ref_model_mixup_alpha=0.6, ref_model_sync_steps=512, top_entropy_quantile=1.0, max_tool_calling_iterations=None, vllm_importance_sampling_correction=True, vllm_importance_sampling_mode='sequence_mask', vllm_importance_sampling_cap=3.0, off_policy_mask_threshold=None, use_bias_correction_kl=False, log_completions=True, num_completions_to_print=None, log_unique_prompts=False, log_completions_hub_repo=None, tuner_backend='peft', vit_gradient_checkpointing=True, router_aux_loss_coef=0.0, enable_dft_loss=False, enable_channel_loss=False, safe_serialization=True, max_shard_size='5GB', check_model=True, acc_strategy='token', train_dataloader_shuffle=True, max_epochs=None, aligner_lr=None, vit_lr=None, use_logits_to_keep=None, resume_only_model=False, optimizer=None, eval_metric=None, callbacks=[], early_stop_interval=None, eval_use_evalscope=False, eval_dataset=[], eval_dataset_args=None, eval_limit=None, eval_generation_config=None, extra_eval_args=None, tuner_type='lora', use_galore=False, galore_target_modules=None, galore_rank=128, galore_update_proj_gap=50, galore_scale=1.0, galore_proj_type='std', galore_optim_per_parameter=False, galore_with_embedding=False, galore_quantization=False, galore_proj_quant=False, galore_proj_bits=4, galore_proj_group_size=256, galore_cos_threshold=0.4, galore_gamma_proj=2, galore_queue_size=5, lisa_activated_layers=0, lisa_step_interval=20, use_flash_ckpt=False, vllm_pipeline_parallel_size=1, vllm_enable_expert_parallel=False, vllm_max_num_seqs=None, vllm_max_model_len=None, vllm_disable_custom_all_reduce=True, vllm_enforce_eager=False, vllm_limit_mm_per_prompt=None, vllm_max_lora_rank=16, vllm_enable_prefix_caching=True, vllm_use_async_engine=None, vllm_quantization=None, vllm_reasoning_parser=None, vllm_disable_cascade_attn=False, vllm_mm_processor_cache_gb=None, vllm_speculative_config=None, vllm_engine_kwargs={}, vllm_data_parallel_size=1, stop_words=[], vllm_enable_lora=False, lora_rank=8, vllm_server_group_port=None, enable_flattened_weight_sync=True, async_generate=False, structured_outputs_regex=None, sleep_level=0, move_model_batches=None, offload_optimizer=False, offload_model=False, wandb_log_unique_prompts=None, cosine_min_len_value_wrong=-0.5, cosine_max_len_value_wrong=0.0, cosine_min_len_value_correct=1.0, cosine_max_len_value_correct=0.5, cosine_max_len=256, repetition_n_grams=3, repetition_max_penalty=-1.0, reward_model=None, reward_model_plugin=None, chord_sft_dataset=[], chord_sft_per_device_train_batch_size=None, chord_enable_phi_function=False, chord_mu_warmup_steps=None, chord_mu_decay_steps=None, chord_mu_peak=None, chord_mu_valley=None, multi_turn_scheduler=None, max_turns=None, completion_length_limit_scope='per_round', vllm_server_pass_dataset=False, dynamic_sample=True, max_resample_times=4, overlong_filter=True, soft_max_length=None, soft_cache_length=None, log_entropy=False, tau_pos=1.0, tau_neg=1.05, advantage_estimator='grpo', kl_in_reward=False, dataset_shuffle=True, rollout_importance_sampling_mode=None, rollout_importance_sampling_threshold=2.0, log_rollout_offpolicy_metrics=False, off_policy_sequence_mask_delta=None)"
+}

lora/lora-stage3/optimizer.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:226e26b176c37ed7c792adfd2fe4136f95d2fdb572f9bb695f787161e8da0faa
+size 99183201

lora/lora-stage3/rng_state_0.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e6aa29f654dcff45f4d494e85fba95c300e2ba77360edeca5a3899f79909e7ce
+size 14725

lora/lora-stage3/rng_state_1.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:47db367bb33a2abe8e3e662eec69e0be4925b4a0a64b5b6c12647bc9faa62ad2
+size 14661

lora/lora-stage3/rng_state_2.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:2faaa8f2708f53af7418ce42a1a06c28bcd4f75dce65c528b4f754d02132f5c0
+size 14661

lora/lora-stage3/scheduler.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:34c1c30cb5e25ddb67ea8d805e6a5f129c70e970f08b80a6781c3286db43ea15
+size 1465

lora/lora-stage3/trainer_state.json ADDED Viewed

The diff for this file is too large to render. See raw diff