arain commited on
Commit
0e94805
·
1 Parent(s): 67bb8fd

upload model files

Browse files
.DS_Store ADDED
Binary file (6.15 kB). View file
 
generation_config.json ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {
2
+ "_from_model_config": true,
3
+ "bos_token_id": 100000,
4
+ "eos_token_id": 100015,
5
+ "transformers_version": "4.34.1"
6
+ }
latest ADDED
@@ -0,0 +1 @@
 
 
1
+ global_step1000
special_tokens_map.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token": {
3
+ "content": "<|begin▁of▁sentence|>",
4
+ "lstrip": false,
5
+ "normalized": false,
6
+ "rstrip": false,
7
+ "single_word": false
8
+ },
9
+ "eos_token": {
10
+ "content": "<|end▁of▁sentence|>",
11
+ "lstrip": false,
12
+ "normalized": false,
13
+ "rstrip": false,
14
+ "single_word": false
15
+ },
16
+ "pad_token": {
17
+ "content": "<|end▁of▁sentence|>",
18
+ "lstrip": false,
19
+ "normalized": true,
20
+ "rstrip": false,
21
+ "single_word": false
22
+ }
23
+ }
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer_config.json ADDED
@@ -0,0 +1,144 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "added_tokens_decoder": {
3
+ "100000": {
4
+ "content": "<|begin▁of▁sentence|>",
5
+ "lstrip": false,
6
+ "normalized": false,
7
+ "rstrip": false,
8
+ "single_word": false,
9
+ "special": true
10
+ },
11
+ "100001": {
12
+ "content": "<|end▁of▁sentence|>",
13
+ "lstrip": false,
14
+ "normalized": false,
15
+ "rstrip": false,
16
+ "single_word": false,
17
+ "special": true
18
+ },
19
+ "100002": {
20
+ "content": "ø",
21
+ "lstrip": false,
22
+ "normalized": true,
23
+ "rstrip": false,
24
+ "single_word": false,
25
+ "special": false
26
+ },
27
+ "100003": {
28
+ "content": "ö",
29
+ "lstrip": false,
30
+ "normalized": true,
31
+ "rstrip": false,
32
+ "single_word": false,
33
+ "special": false
34
+ },
35
+ "100004": {
36
+ "content": "ú",
37
+ "lstrip": false,
38
+ "normalized": true,
39
+ "rstrip": false,
40
+ "single_word": false,
41
+ "special": false
42
+ },
43
+ "100005": {
44
+ "content": "ÿ",
45
+ "lstrip": false,
46
+ "normalized": true,
47
+ "rstrip": false,
48
+ "single_word": false,
49
+ "special": false
50
+ },
51
+ "100006": {
52
+ "content": "õ",
53
+ "lstrip": false,
54
+ "normalized": true,
55
+ "rstrip": false,
56
+ "single_word": false,
57
+ "special": false
58
+ },
59
+ "100007": {
60
+ "content": "÷",
61
+ "lstrip": false,
62
+ "normalized": true,
63
+ "rstrip": false,
64
+ "single_word": false,
65
+ "special": false
66
+ },
67
+ "100008": {
68
+ "content": "û",
69
+ "lstrip": false,
70
+ "normalized": true,
71
+ "rstrip": false,
72
+ "single_word": false,
73
+ "special": false
74
+ },
75
+ "100009": {
76
+ "content": "ý",
77
+ "lstrip": false,
78
+ "normalized": true,
79
+ "rstrip": false,
80
+ "single_word": false,
81
+ "special": false
82
+ },
83
+ "100010": {
84
+ "content": "À",
85
+ "lstrip": false,
86
+ "normalized": true,
87
+ "rstrip": false,
88
+ "single_word": false,
89
+ "special": false
90
+ },
91
+ "100011": {
92
+ "content": "ù",
93
+ "lstrip": false,
94
+ "normalized": true,
95
+ "rstrip": false,
96
+ "single_word": false,
97
+ "special": false
98
+ },
99
+ "100012": {
100
+ "content": "Á",
101
+ "lstrip": false,
102
+ "normalized": true,
103
+ "rstrip": false,
104
+ "single_word": false,
105
+ "special": false
106
+ },
107
+ "100013": {
108
+ "content": "þ",
109
+ "lstrip": false,
110
+ "normalized": true,
111
+ "rstrip": false,
112
+ "single_word": false,
113
+ "special": false
114
+ },
115
+ "100014": {
116
+ "content": "ü",
117
+ "lstrip": false,
118
+ "normalized": true,
119
+ "rstrip": false,
120
+ "single_word": false,
121
+ "special": false
122
+ },
123
+ "100015": {
124
+ "content": "<|EOT|>",
125
+ "lstrip": false,
126
+ "normalized": true,
127
+ "rstrip": false,
128
+ "single_word": false,
129
+ "special": true
130
+ }
131
+ },
132
+ "bos_token": "<|begin▁of▁sentence|>",
133
+ "chat_template": "{% if not add_generation_prompt is defined %}\n{% set add_generation_prompt = false %}\n{% endif %}\n{%- set ns = namespace(found=false) -%}\n{%- for message in messages -%}\n {%- if message['role'] == 'system' -%}\n {%- set ns.found = true -%}\n {%- endif -%}\n{%- endfor -%}\n{{bos_token}}{%- if not ns.found -%}\n{{'You are an AI programming assistant, utilizing the Deepseek Coder model, developed by Deepseek Company, and you only answer questions related to computer science. For politically sensitive questions, security and privacy issues, and other non-computer science questions, you will refuse to answer\\n'}}\n{%- endif %}\n{%- for message in messages %}\n {%- if message['role'] == 'system' %}\n{{ message['content'] }}\n {%- else %}\n {%- if message['role'] == 'user' %}\n{{'### Instruction:\\n' + message['content'] + '\\n'}}\n {%- else %}\n{{'### Response:\\n' + message['content'] + '\\n<|EOT|>\\n'}}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{% if add_generation_prompt %}\n{{'### Response:'}}\n{% endif %}",
134
+ "clean_up_tokenization_spaces": false,
135
+ "eos_token": "<|end▁of▁sentence|>",
136
+ "legacy": true,
137
+ "model_max_length": 4096,
138
+ "pad_token": "<|end▁of▁sentence|>",
139
+ "padding_side": "right",
140
+ "sp_model_kwargs": {},
141
+ "tokenizer_class": "LlamaTokenizer",
142
+ "unk_token": null,
143
+ "use_default_system_prompt": true
144
+ }
trainer_state.json ADDED
@@ -0,0 +1,3019 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_metric": null,
3
+ "best_model_checkpoint": null,
4
+ "epoch": 3.097173828881146,
5
+ "eval_steps": 500,
6
+ "global_step": 1000,
7
+ "is_hyper_param_search": false,
8
+ "is_local_process_zero": true,
9
+ "is_world_process_zero": true,
10
+ "log_history": [
11
+ {
12
+ "epoch": 0.01,
13
+ "learning_rate": 6.020599913279623e-06,
14
+ "loss": 1.2605,
15
+ "step": 2
16
+ },
17
+ {
18
+ "epoch": 0.01,
19
+ "learning_rate": 1.2041199826559246e-05,
20
+ "loss": 1.1266,
21
+ "step": 4
22
+ },
23
+ {
24
+ "epoch": 0.02,
25
+ "learning_rate": 1.5563025007672873e-05,
26
+ "loss": 0.9129,
27
+ "step": 6
28
+ },
29
+ {
30
+ "epoch": 0.02,
31
+ "learning_rate": 1.806179973983887e-05,
32
+ "loss": 0.7453,
33
+ "step": 8
34
+ },
35
+ {
36
+ "epoch": 0.03,
37
+ "learning_rate": 1.9999999999999998e-05,
38
+ "loss": 0.6517,
39
+ "step": 10
40
+ },
41
+ {
42
+ "epoch": 0.04,
43
+ "learning_rate": 2e-05,
44
+ "loss": 0.5941,
45
+ "step": 12
46
+ },
47
+ {
48
+ "epoch": 0.04,
49
+ "learning_rate": 2e-05,
50
+ "loss": 0.5568,
51
+ "step": 14
52
+ },
53
+ {
54
+ "epoch": 0.05,
55
+ "learning_rate": 2e-05,
56
+ "loss": 0.5374,
57
+ "step": 16
58
+ },
59
+ {
60
+ "epoch": 0.06,
61
+ "learning_rate": 2e-05,
62
+ "loss": 0.5121,
63
+ "step": 18
64
+ },
65
+ {
66
+ "epoch": 0.06,
67
+ "learning_rate": 2e-05,
68
+ "loss": 0.5023,
69
+ "step": 20
70
+ },
71
+ {
72
+ "epoch": 0.07,
73
+ "learning_rate": 2e-05,
74
+ "loss": 0.4889,
75
+ "step": 22
76
+ },
77
+ {
78
+ "epoch": 0.07,
79
+ "learning_rate": 2e-05,
80
+ "loss": 0.4846,
81
+ "step": 24
82
+ },
83
+ {
84
+ "epoch": 0.08,
85
+ "learning_rate": 2e-05,
86
+ "loss": 0.4739,
87
+ "step": 26
88
+ },
89
+ {
90
+ "epoch": 0.09,
91
+ "learning_rate": 2e-05,
92
+ "loss": 0.4667,
93
+ "step": 28
94
+ },
95
+ {
96
+ "epoch": 0.09,
97
+ "learning_rate": 2e-05,
98
+ "loss": 0.4566,
99
+ "step": 30
100
+ },
101
+ {
102
+ "epoch": 0.1,
103
+ "learning_rate": 2e-05,
104
+ "loss": 0.4544,
105
+ "step": 32
106
+ },
107
+ {
108
+ "epoch": 0.11,
109
+ "learning_rate": 2e-05,
110
+ "loss": 0.4409,
111
+ "step": 34
112
+ },
113
+ {
114
+ "epoch": 0.11,
115
+ "learning_rate": 2e-05,
116
+ "loss": 0.4397,
117
+ "step": 36
118
+ },
119
+ {
120
+ "epoch": 0.12,
121
+ "learning_rate": 2e-05,
122
+ "loss": 0.4408,
123
+ "step": 38
124
+ },
125
+ {
126
+ "epoch": 0.12,
127
+ "learning_rate": 2e-05,
128
+ "loss": 0.4376,
129
+ "step": 40
130
+ },
131
+ {
132
+ "epoch": 0.13,
133
+ "learning_rate": 2e-05,
134
+ "loss": 0.4279,
135
+ "step": 42
136
+ },
137
+ {
138
+ "epoch": 0.14,
139
+ "learning_rate": 2e-05,
140
+ "loss": 0.4267,
141
+ "step": 44
142
+ },
143
+ {
144
+ "epoch": 0.14,
145
+ "learning_rate": 2e-05,
146
+ "loss": 0.4209,
147
+ "step": 46
148
+ },
149
+ {
150
+ "epoch": 0.15,
151
+ "learning_rate": 2e-05,
152
+ "loss": 0.4169,
153
+ "step": 48
154
+ },
155
+ {
156
+ "epoch": 0.15,
157
+ "learning_rate": 2e-05,
158
+ "loss": 0.4171,
159
+ "step": 50
160
+ },
161
+ {
162
+ "epoch": 0.16,
163
+ "learning_rate": 2e-05,
164
+ "loss": 0.4162,
165
+ "step": 52
166
+ },
167
+ {
168
+ "epoch": 0.17,
169
+ "learning_rate": 2e-05,
170
+ "loss": 0.4187,
171
+ "step": 54
172
+ },
173
+ {
174
+ "epoch": 0.17,
175
+ "learning_rate": 2e-05,
176
+ "loss": 0.4138,
177
+ "step": 56
178
+ },
179
+ {
180
+ "epoch": 0.18,
181
+ "learning_rate": 2e-05,
182
+ "loss": 0.4184,
183
+ "step": 58
184
+ },
185
+ {
186
+ "epoch": 0.19,
187
+ "learning_rate": 2e-05,
188
+ "loss": 0.4118,
189
+ "step": 60
190
+ },
191
+ {
192
+ "epoch": 0.19,
193
+ "learning_rate": 2e-05,
194
+ "loss": 0.4039,
195
+ "step": 62
196
+ },
197
+ {
198
+ "epoch": 0.2,
199
+ "learning_rate": 2e-05,
200
+ "loss": 0.404,
201
+ "step": 64
202
+ },
203
+ {
204
+ "epoch": 0.2,
205
+ "learning_rate": 2e-05,
206
+ "loss": 0.4031,
207
+ "step": 66
208
+ },
209
+ {
210
+ "epoch": 0.21,
211
+ "learning_rate": 2e-05,
212
+ "loss": 0.3959,
213
+ "step": 68
214
+ },
215
+ {
216
+ "epoch": 0.22,
217
+ "learning_rate": 2e-05,
218
+ "loss": 0.4046,
219
+ "step": 70
220
+ },
221
+ {
222
+ "epoch": 0.22,
223
+ "learning_rate": 2e-05,
224
+ "loss": 0.3943,
225
+ "step": 72
226
+ },
227
+ {
228
+ "epoch": 0.23,
229
+ "learning_rate": 2e-05,
230
+ "loss": 0.3995,
231
+ "step": 74
232
+ },
233
+ {
234
+ "epoch": 0.24,
235
+ "learning_rate": 2e-05,
236
+ "loss": 0.3925,
237
+ "step": 76
238
+ },
239
+ {
240
+ "epoch": 0.24,
241
+ "learning_rate": 2e-05,
242
+ "loss": 0.3908,
243
+ "step": 78
244
+ },
245
+ {
246
+ "epoch": 0.25,
247
+ "learning_rate": 2e-05,
248
+ "loss": 0.3901,
249
+ "step": 80
250
+ },
251
+ {
252
+ "epoch": 0.25,
253
+ "learning_rate": 2e-05,
254
+ "loss": 0.3848,
255
+ "step": 82
256
+ },
257
+ {
258
+ "epoch": 0.26,
259
+ "learning_rate": 2e-05,
260
+ "loss": 0.3877,
261
+ "step": 84
262
+ },
263
+ {
264
+ "epoch": 0.27,
265
+ "learning_rate": 2e-05,
266
+ "loss": 0.3857,
267
+ "step": 86
268
+ },
269
+ {
270
+ "epoch": 0.27,
271
+ "learning_rate": 2e-05,
272
+ "loss": 0.381,
273
+ "step": 88
274
+ },
275
+ {
276
+ "epoch": 0.28,
277
+ "learning_rate": 2e-05,
278
+ "loss": 0.3841,
279
+ "step": 90
280
+ },
281
+ {
282
+ "epoch": 0.28,
283
+ "learning_rate": 2e-05,
284
+ "loss": 0.3895,
285
+ "step": 92
286
+ },
287
+ {
288
+ "epoch": 0.29,
289
+ "learning_rate": 2e-05,
290
+ "loss": 0.3774,
291
+ "step": 94
292
+ },
293
+ {
294
+ "epoch": 0.3,
295
+ "learning_rate": 2e-05,
296
+ "loss": 0.3794,
297
+ "step": 96
298
+ },
299
+ {
300
+ "epoch": 0.3,
301
+ "learning_rate": 2e-05,
302
+ "loss": 0.3761,
303
+ "step": 98
304
+ },
305
+ {
306
+ "epoch": 0.31,
307
+ "learning_rate": 2e-05,
308
+ "loss": 0.376,
309
+ "step": 100
310
+ },
311
+ {
312
+ "epoch": 0.32,
313
+ "learning_rate": 2e-05,
314
+ "loss": 0.3808,
315
+ "step": 102
316
+ },
317
+ {
318
+ "epoch": 0.32,
319
+ "learning_rate": 2e-05,
320
+ "loss": 0.3791,
321
+ "step": 104
322
+ },
323
+ {
324
+ "epoch": 0.33,
325
+ "learning_rate": 2e-05,
326
+ "loss": 0.3716,
327
+ "step": 106
328
+ },
329
+ {
330
+ "epoch": 0.33,
331
+ "learning_rate": 2e-05,
332
+ "loss": 0.377,
333
+ "step": 108
334
+ },
335
+ {
336
+ "epoch": 0.34,
337
+ "learning_rate": 2e-05,
338
+ "loss": 0.3684,
339
+ "step": 110
340
+ },
341
+ {
342
+ "epoch": 0.35,
343
+ "learning_rate": 2e-05,
344
+ "loss": 0.3665,
345
+ "step": 112
346
+ },
347
+ {
348
+ "epoch": 0.35,
349
+ "learning_rate": 2e-05,
350
+ "loss": 0.3705,
351
+ "step": 114
352
+ },
353
+ {
354
+ "epoch": 0.36,
355
+ "learning_rate": 2e-05,
356
+ "loss": 0.3719,
357
+ "step": 116
358
+ },
359
+ {
360
+ "epoch": 0.37,
361
+ "learning_rate": 2e-05,
362
+ "loss": 0.371,
363
+ "step": 118
364
+ },
365
+ {
366
+ "epoch": 0.37,
367
+ "learning_rate": 2e-05,
368
+ "loss": 0.3686,
369
+ "step": 120
370
+ },
371
+ {
372
+ "epoch": 0.38,
373
+ "learning_rate": 2e-05,
374
+ "loss": 0.3689,
375
+ "step": 122
376
+ },
377
+ {
378
+ "epoch": 0.38,
379
+ "learning_rate": 2e-05,
380
+ "loss": 0.3519,
381
+ "step": 124
382
+ },
383
+ {
384
+ "epoch": 0.39,
385
+ "learning_rate": 2e-05,
386
+ "loss": 0.3601,
387
+ "step": 126
388
+ },
389
+ {
390
+ "epoch": 0.4,
391
+ "learning_rate": 2e-05,
392
+ "loss": 0.3615,
393
+ "step": 128
394
+ },
395
+ {
396
+ "epoch": 0.4,
397
+ "learning_rate": 2e-05,
398
+ "loss": 0.3602,
399
+ "step": 130
400
+ },
401
+ {
402
+ "epoch": 0.41,
403
+ "learning_rate": 2e-05,
404
+ "loss": 0.3618,
405
+ "step": 132
406
+ },
407
+ {
408
+ "epoch": 0.42,
409
+ "learning_rate": 2e-05,
410
+ "loss": 0.36,
411
+ "step": 134
412
+ },
413
+ {
414
+ "epoch": 0.42,
415
+ "learning_rate": 2e-05,
416
+ "loss": 0.3538,
417
+ "step": 136
418
+ },
419
+ {
420
+ "epoch": 0.43,
421
+ "learning_rate": 2e-05,
422
+ "loss": 0.3534,
423
+ "step": 138
424
+ },
425
+ {
426
+ "epoch": 0.43,
427
+ "learning_rate": 2e-05,
428
+ "loss": 0.3651,
429
+ "step": 140
430
+ },
431
+ {
432
+ "epoch": 0.44,
433
+ "learning_rate": 2e-05,
434
+ "loss": 0.3543,
435
+ "step": 142
436
+ },
437
+ {
438
+ "epoch": 0.45,
439
+ "learning_rate": 2e-05,
440
+ "loss": 0.3521,
441
+ "step": 144
442
+ },
443
+ {
444
+ "epoch": 0.45,
445
+ "learning_rate": 2e-05,
446
+ "loss": 0.3503,
447
+ "step": 146
448
+ },
449
+ {
450
+ "epoch": 0.46,
451
+ "learning_rate": 2e-05,
452
+ "loss": 0.3443,
453
+ "step": 148
454
+ },
455
+ {
456
+ "epoch": 0.46,
457
+ "learning_rate": 2e-05,
458
+ "loss": 0.3486,
459
+ "step": 150
460
+ },
461
+ {
462
+ "epoch": 0.47,
463
+ "learning_rate": 2e-05,
464
+ "loss": 0.348,
465
+ "step": 152
466
+ },
467
+ {
468
+ "epoch": 0.48,
469
+ "learning_rate": 2e-05,
470
+ "loss": 0.3509,
471
+ "step": 154
472
+ },
473
+ {
474
+ "epoch": 0.48,
475
+ "learning_rate": 2e-05,
476
+ "loss": 0.3406,
477
+ "step": 156
478
+ },
479
+ {
480
+ "epoch": 0.49,
481
+ "learning_rate": 2e-05,
482
+ "loss": 0.3476,
483
+ "step": 158
484
+ },
485
+ {
486
+ "epoch": 0.5,
487
+ "learning_rate": 2e-05,
488
+ "loss": 0.3385,
489
+ "step": 160
490
+ },
491
+ {
492
+ "epoch": 0.5,
493
+ "learning_rate": 2e-05,
494
+ "loss": 0.3419,
495
+ "step": 162
496
+ },
497
+ {
498
+ "epoch": 0.51,
499
+ "learning_rate": 2e-05,
500
+ "loss": 0.3449,
501
+ "step": 164
502
+ },
503
+ {
504
+ "epoch": 0.51,
505
+ "learning_rate": 2e-05,
506
+ "loss": 0.335,
507
+ "step": 166
508
+ },
509
+ {
510
+ "epoch": 0.52,
511
+ "learning_rate": 2e-05,
512
+ "loss": 0.3395,
513
+ "step": 168
514
+ },
515
+ {
516
+ "epoch": 0.53,
517
+ "learning_rate": 2e-05,
518
+ "loss": 0.3364,
519
+ "step": 170
520
+ },
521
+ {
522
+ "epoch": 0.53,
523
+ "learning_rate": 2e-05,
524
+ "loss": 0.3377,
525
+ "step": 172
526
+ },
527
+ {
528
+ "epoch": 0.54,
529
+ "learning_rate": 2e-05,
530
+ "loss": 0.339,
531
+ "step": 174
532
+ },
533
+ {
534
+ "epoch": 0.55,
535
+ "learning_rate": 2e-05,
536
+ "loss": 0.3358,
537
+ "step": 176
538
+ },
539
+ {
540
+ "epoch": 0.55,
541
+ "learning_rate": 2e-05,
542
+ "loss": 0.3402,
543
+ "step": 178
544
+ },
545
+ {
546
+ "epoch": 0.56,
547
+ "learning_rate": 2e-05,
548
+ "loss": 0.3324,
549
+ "step": 180
550
+ },
551
+ {
552
+ "epoch": 0.56,
553
+ "learning_rate": 2e-05,
554
+ "loss": 0.3363,
555
+ "step": 182
556
+ },
557
+ {
558
+ "epoch": 0.57,
559
+ "learning_rate": 2e-05,
560
+ "loss": 0.3329,
561
+ "step": 184
562
+ },
563
+ {
564
+ "epoch": 0.58,
565
+ "learning_rate": 2e-05,
566
+ "loss": 0.3282,
567
+ "step": 186
568
+ },
569
+ {
570
+ "epoch": 0.58,
571
+ "learning_rate": 2e-05,
572
+ "loss": 0.3363,
573
+ "step": 188
574
+ },
575
+ {
576
+ "epoch": 0.59,
577
+ "learning_rate": 2e-05,
578
+ "loss": 0.3334,
579
+ "step": 190
580
+ },
581
+ {
582
+ "epoch": 0.59,
583
+ "learning_rate": 2e-05,
584
+ "loss": 0.331,
585
+ "step": 192
586
+ },
587
+ {
588
+ "epoch": 0.6,
589
+ "learning_rate": 2e-05,
590
+ "loss": 0.3297,
591
+ "step": 194
592
+ },
593
+ {
594
+ "epoch": 0.61,
595
+ "learning_rate": 2e-05,
596
+ "loss": 0.331,
597
+ "step": 196
598
+ },
599
+ {
600
+ "epoch": 0.61,
601
+ "learning_rate": 2e-05,
602
+ "loss": 0.3277,
603
+ "step": 198
604
+ },
605
+ {
606
+ "epoch": 0.62,
607
+ "learning_rate": 2e-05,
608
+ "loss": 0.3246,
609
+ "step": 200
610
+ },
611
+ {
612
+ "epoch": 0.63,
613
+ "learning_rate": 2e-05,
614
+ "loss": 0.3269,
615
+ "step": 202
616
+ },
617
+ {
618
+ "epoch": 0.63,
619
+ "learning_rate": 2e-05,
620
+ "loss": 0.3269,
621
+ "step": 204
622
+ },
623
+ {
624
+ "epoch": 0.64,
625
+ "learning_rate": 2e-05,
626
+ "loss": 0.3261,
627
+ "step": 206
628
+ },
629
+ {
630
+ "epoch": 0.64,
631
+ "learning_rate": 2e-05,
632
+ "loss": 0.3288,
633
+ "step": 208
634
+ },
635
+ {
636
+ "epoch": 0.65,
637
+ "learning_rate": 2e-05,
638
+ "loss": 0.3266,
639
+ "step": 210
640
+ },
641
+ {
642
+ "epoch": 0.66,
643
+ "learning_rate": 2e-05,
644
+ "loss": 0.3239,
645
+ "step": 212
646
+ },
647
+ {
648
+ "epoch": 0.66,
649
+ "learning_rate": 2e-05,
650
+ "loss": 0.3247,
651
+ "step": 214
652
+ },
653
+ {
654
+ "epoch": 0.67,
655
+ "learning_rate": 2e-05,
656
+ "loss": 0.3215,
657
+ "step": 216
658
+ },
659
+ {
660
+ "epoch": 0.68,
661
+ "learning_rate": 2e-05,
662
+ "loss": 0.3212,
663
+ "step": 218
664
+ },
665
+ {
666
+ "epoch": 0.68,
667
+ "learning_rate": 2e-05,
668
+ "loss": 0.3195,
669
+ "step": 220
670
+ },
671
+ {
672
+ "epoch": 0.69,
673
+ "learning_rate": 2e-05,
674
+ "loss": 0.3187,
675
+ "step": 222
676
+ },
677
+ {
678
+ "epoch": 0.69,
679
+ "learning_rate": 2e-05,
680
+ "loss": 0.3179,
681
+ "step": 224
682
+ },
683
+ {
684
+ "epoch": 0.7,
685
+ "learning_rate": 2e-05,
686
+ "loss": 0.3159,
687
+ "step": 226
688
+ },
689
+ {
690
+ "epoch": 0.71,
691
+ "learning_rate": 2e-05,
692
+ "loss": 0.3193,
693
+ "step": 228
694
+ },
695
+ {
696
+ "epoch": 0.71,
697
+ "learning_rate": 2e-05,
698
+ "loss": 0.3168,
699
+ "step": 230
700
+ },
701
+ {
702
+ "epoch": 0.72,
703
+ "learning_rate": 2e-05,
704
+ "loss": 0.3164,
705
+ "step": 232
706
+ },
707
+ {
708
+ "epoch": 0.72,
709
+ "learning_rate": 2e-05,
710
+ "loss": 0.3174,
711
+ "step": 234
712
+ },
713
+ {
714
+ "epoch": 0.73,
715
+ "learning_rate": 2e-05,
716
+ "loss": 0.3185,
717
+ "step": 236
718
+ },
719
+ {
720
+ "epoch": 0.74,
721
+ "learning_rate": 2e-05,
722
+ "loss": 0.3173,
723
+ "step": 238
724
+ },
725
+ {
726
+ "epoch": 0.74,
727
+ "learning_rate": 2e-05,
728
+ "loss": 0.3132,
729
+ "step": 240
730
+ },
731
+ {
732
+ "epoch": 0.75,
733
+ "learning_rate": 2e-05,
734
+ "loss": 0.3116,
735
+ "step": 242
736
+ },
737
+ {
738
+ "epoch": 0.76,
739
+ "learning_rate": 2e-05,
740
+ "loss": 0.3142,
741
+ "step": 244
742
+ },
743
+ {
744
+ "epoch": 0.76,
745
+ "learning_rate": 2e-05,
746
+ "loss": 0.3086,
747
+ "step": 246
748
+ },
749
+ {
750
+ "epoch": 0.77,
751
+ "learning_rate": 2e-05,
752
+ "loss": 0.3154,
753
+ "step": 248
754
+ },
755
+ {
756
+ "epoch": 0.77,
757
+ "learning_rate": 2e-05,
758
+ "loss": 0.3053,
759
+ "step": 250
760
+ },
761
+ {
762
+ "epoch": 0.78,
763
+ "learning_rate": 2e-05,
764
+ "loss": 0.3108,
765
+ "step": 252
766
+ },
767
+ {
768
+ "epoch": 0.79,
769
+ "learning_rate": 2e-05,
770
+ "loss": 0.3081,
771
+ "step": 254
772
+ },
773
+ {
774
+ "epoch": 0.79,
775
+ "learning_rate": 2e-05,
776
+ "loss": 0.3041,
777
+ "step": 256
778
+ },
779
+ {
780
+ "epoch": 0.8,
781
+ "learning_rate": 2e-05,
782
+ "loss": 0.3112,
783
+ "step": 258
784
+ },
785
+ {
786
+ "epoch": 0.81,
787
+ "learning_rate": 2e-05,
788
+ "loss": 0.3077,
789
+ "step": 260
790
+ },
791
+ {
792
+ "epoch": 0.81,
793
+ "learning_rate": 2e-05,
794
+ "loss": 0.3032,
795
+ "step": 262
796
+ },
797
+ {
798
+ "epoch": 0.82,
799
+ "learning_rate": 2e-05,
800
+ "loss": 0.3068,
801
+ "step": 264
802
+ },
803
+ {
804
+ "epoch": 0.82,
805
+ "learning_rate": 2e-05,
806
+ "loss": 0.3111,
807
+ "step": 266
808
+ },
809
+ {
810
+ "epoch": 0.83,
811
+ "learning_rate": 2e-05,
812
+ "loss": 0.3061,
813
+ "step": 268
814
+ },
815
+ {
816
+ "epoch": 0.84,
817
+ "learning_rate": 2e-05,
818
+ "loss": 0.3015,
819
+ "step": 270
820
+ },
821
+ {
822
+ "epoch": 0.84,
823
+ "learning_rate": 2e-05,
824
+ "loss": 0.3043,
825
+ "step": 272
826
+ },
827
+ {
828
+ "epoch": 0.85,
829
+ "learning_rate": 2e-05,
830
+ "loss": 0.2996,
831
+ "step": 274
832
+ },
833
+ {
834
+ "epoch": 0.85,
835
+ "learning_rate": 2e-05,
836
+ "loss": 0.3047,
837
+ "step": 276
838
+ },
839
+ {
840
+ "epoch": 0.86,
841
+ "learning_rate": 2e-05,
842
+ "loss": 0.2996,
843
+ "step": 278
844
+ },
845
+ {
846
+ "epoch": 0.87,
847
+ "learning_rate": 2e-05,
848
+ "loss": 0.3038,
849
+ "step": 280
850
+ },
851
+ {
852
+ "epoch": 0.87,
853
+ "learning_rate": 2e-05,
854
+ "loss": 0.2977,
855
+ "step": 282
856
+ },
857
+ {
858
+ "epoch": 0.88,
859
+ "learning_rate": 2e-05,
860
+ "loss": 0.2981,
861
+ "step": 284
862
+ },
863
+ {
864
+ "epoch": 0.89,
865
+ "learning_rate": 2e-05,
866
+ "loss": 0.3054,
867
+ "step": 286
868
+ },
869
+ {
870
+ "epoch": 0.89,
871
+ "learning_rate": 2e-05,
872
+ "loss": 0.3002,
873
+ "step": 288
874
+ },
875
+ {
876
+ "epoch": 0.9,
877
+ "learning_rate": 2e-05,
878
+ "loss": 0.3028,
879
+ "step": 290
880
+ },
881
+ {
882
+ "epoch": 0.9,
883
+ "learning_rate": 2e-05,
884
+ "loss": 0.2942,
885
+ "step": 292
886
+ },
887
+ {
888
+ "epoch": 0.91,
889
+ "learning_rate": 2e-05,
890
+ "loss": 0.2966,
891
+ "step": 294
892
+ },
893
+ {
894
+ "epoch": 0.92,
895
+ "learning_rate": 2e-05,
896
+ "loss": 0.2955,
897
+ "step": 296
898
+ },
899
+ {
900
+ "epoch": 0.92,
901
+ "learning_rate": 2e-05,
902
+ "loss": 0.3015,
903
+ "step": 298
904
+ },
905
+ {
906
+ "epoch": 0.93,
907
+ "learning_rate": 2e-05,
908
+ "loss": 0.2933,
909
+ "step": 300
910
+ },
911
+ {
912
+ "epoch": 0.94,
913
+ "learning_rate": 2e-05,
914
+ "loss": 0.2935,
915
+ "step": 302
916
+ },
917
+ {
918
+ "epoch": 0.94,
919
+ "learning_rate": 2e-05,
920
+ "loss": 0.2979,
921
+ "step": 304
922
+ },
923
+ {
924
+ "epoch": 0.95,
925
+ "learning_rate": 2e-05,
926
+ "loss": 0.2951,
927
+ "step": 306
928
+ },
929
+ {
930
+ "epoch": 0.95,
931
+ "learning_rate": 2e-05,
932
+ "loss": 0.2901,
933
+ "step": 308
934
+ },
935
+ {
936
+ "epoch": 0.96,
937
+ "learning_rate": 2e-05,
938
+ "loss": 0.2885,
939
+ "step": 310
940
+ },
941
+ {
942
+ "epoch": 0.97,
943
+ "learning_rate": 2e-05,
944
+ "loss": 0.2897,
945
+ "step": 312
946
+ },
947
+ {
948
+ "epoch": 0.97,
949
+ "learning_rate": 2e-05,
950
+ "loss": 0.291,
951
+ "step": 314
952
+ },
953
+ {
954
+ "epoch": 0.98,
955
+ "learning_rate": 2e-05,
956
+ "loss": 0.2919,
957
+ "step": 316
958
+ },
959
+ {
960
+ "epoch": 0.98,
961
+ "learning_rate": 2e-05,
962
+ "loss": 0.2869,
963
+ "step": 318
964
+ },
965
+ {
966
+ "epoch": 0.99,
967
+ "learning_rate": 2e-05,
968
+ "loss": 0.2873,
969
+ "step": 320
970
+ },
971
+ {
972
+ "epoch": 1.0,
973
+ "learning_rate": 2e-05,
974
+ "loss": 0.2913,
975
+ "step": 322
976
+ },
977
+ {
978
+ "epoch": 1.0,
979
+ "learning_rate": 2e-05,
980
+ "loss": 0.2778,
981
+ "step": 324
982
+ },
983
+ {
984
+ "epoch": 1.01,
985
+ "learning_rate": 2e-05,
986
+ "loss": 0.2778,
987
+ "step": 326
988
+ },
989
+ {
990
+ "epoch": 1.02,
991
+ "learning_rate": 2e-05,
992
+ "loss": 0.2808,
993
+ "step": 328
994
+ },
995
+ {
996
+ "epoch": 1.02,
997
+ "learning_rate": 2e-05,
998
+ "loss": 0.2713,
999
+ "step": 330
1000
+ },
1001
+ {
1002
+ "epoch": 1.03,
1003
+ "learning_rate": 2e-05,
1004
+ "loss": 0.2719,
1005
+ "step": 332
1006
+ },
1007
+ {
1008
+ "epoch": 1.03,
1009
+ "learning_rate": 2e-05,
1010
+ "loss": 0.2749,
1011
+ "step": 334
1012
+ },
1013
+ {
1014
+ "epoch": 1.04,
1015
+ "learning_rate": 2e-05,
1016
+ "loss": 0.269,
1017
+ "step": 336
1018
+ },
1019
+ {
1020
+ "epoch": 1.05,
1021
+ "learning_rate": 2e-05,
1022
+ "loss": 0.272,
1023
+ "step": 338
1024
+ },
1025
+ {
1026
+ "epoch": 1.05,
1027
+ "learning_rate": 2e-05,
1028
+ "loss": 0.2662,
1029
+ "step": 340
1030
+ },
1031
+ {
1032
+ "epoch": 1.06,
1033
+ "learning_rate": 2e-05,
1034
+ "loss": 0.2745,
1035
+ "step": 342
1036
+ },
1037
+ {
1038
+ "epoch": 1.07,
1039
+ "learning_rate": 2e-05,
1040
+ "loss": 0.2697,
1041
+ "step": 344
1042
+ },
1043
+ {
1044
+ "epoch": 1.07,
1045
+ "learning_rate": 2e-05,
1046
+ "loss": 0.2725,
1047
+ "step": 346
1048
+ },
1049
+ {
1050
+ "epoch": 1.08,
1051
+ "learning_rate": 2e-05,
1052
+ "loss": 0.2697,
1053
+ "step": 348
1054
+ },
1055
+ {
1056
+ "epoch": 1.08,
1057
+ "learning_rate": 2e-05,
1058
+ "loss": 0.2783,
1059
+ "step": 350
1060
+ },
1061
+ {
1062
+ "epoch": 1.09,
1063
+ "learning_rate": 2e-05,
1064
+ "loss": 0.2699,
1065
+ "step": 352
1066
+ },
1067
+ {
1068
+ "epoch": 1.1,
1069
+ "learning_rate": 2e-05,
1070
+ "loss": 0.2677,
1071
+ "step": 354
1072
+ },
1073
+ {
1074
+ "epoch": 1.1,
1075
+ "learning_rate": 2e-05,
1076
+ "loss": 0.2667,
1077
+ "step": 356
1078
+ },
1079
+ {
1080
+ "epoch": 1.11,
1081
+ "learning_rate": 2e-05,
1082
+ "loss": 0.2746,
1083
+ "step": 358
1084
+ },
1085
+ {
1086
+ "epoch": 1.11,
1087
+ "learning_rate": 2e-05,
1088
+ "loss": 0.2714,
1089
+ "step": 360
1090
+ },
1091
+ {
1092
+ "epoch": 1.12,
1093
+ "learning_rate": 2e-05,
1094
+ "loss": 0.2677,
1095
+ "step": 362
1096
+ },
1097
+ {
1098
+ "epoch": 1.13,
1099
+ "learning_rate": 2e-05,
1100
+ "loss": 0.2656,
1101
+ "step": 364
1102
+ },
1103
+ {
1104
+ "epoch": 1.13,
1105
+ "learning_rate": 2e-05,
1106
+ "loss": 0.2667,
1107
+ "step": 366
1108
+ },
1109
+ {
1110
+ "epoch": 1.14,
1111
+ "learning_rate": 2e-05,
1112
+ "loss": 0.266,
1113
+ "step": 368
1114
+ },
1115
+ {
1116
+ "epoch": 1.15,
1117
+ "learning_rate": 2e-05,
1118
+ "loss": 0.2691,
1119
+ "step": 370
1120
+ },
1121
+ {
1122
+ "epoch": 1.15,
1123
+ "learning_rate": 2e-05,
1124
+ "loss": 0.2671,
1125
+ "step": 372
1126
+ },
1127
+ {
1128
+ "epoch": 1.16,
1129
+ "learning_rate": 2e-05,
1130
+ "loss": 0.2679,
1131
+ "step": 374
1132
+ },
1133
+ {
1134
+ "epoch": 1.16,
1135
+ "learning_rate": 2e-05,
1136
+ "loss": 0.263,
1137
+ "step": 376
1138
+ },
1139
+ {
1140
+ "epoch": 1.17,
1141
+ "learning_rate": 2e-05,
1142
+ "loss": 0.262,
1143
+ "step": 378
1144
+ },
1145
+ {
1146
+ "epoch": 1.18,
1147
+ "learning_rate": 2e-05,
1148
+ "loss": 0.2668,
1149
+ "step": 380
1150
+ },
1151
+ {
1152
+ "epoch": 1.18,
1153
+ "learning_rate": 2e-05,
1154
+ "loss": 0.265,
1155
+ "step": 382
1156
+ },
1157
+ {
1158
+ "epoch": 1.19,
1159
+ "learning_rate": 2e-05,
1160
+ "loss": 0.2674,
1161
+ "step": 384
1162
+ },
1163
+ {
1164
+ "epoch": 1.2,
1165
+ "learning_rate": 2e-05,
1166
+ "loss": 0.2614,
1167
+ "step": 386
1168
+ },
1169
+ {
1170
+ "epoch": 1.2,
1171
+ "learning_rate": 2e-05,
1172
+ "loss": 0.2612,
1173
+ "step": 388
1174
+ },
1175
+ {
1176
+ "epoch": 1.21,
1177
+ "learning_rate": 2e-05,
1178
+ "loss": 0.2582,
1179
+ "step": 390
1180
+ },
1181
+ {
1182
+ "epoch": 1.21,
1183
+ "learning_rate": 2e-05,
1184
+ "loss": 0.2662,
1185
+ "step": 392
1186
+ },
1187
+ {
1188
+ "epoch": 1.22,
1189
+ "learning_rate": 2e-05,
1190
+ "loss": 0.2652,
1191
+ "step": 394
1192
+ },
1193
+ {
1194
+ "epoch": 1.23,
1195
+ "learning_rate": 2e-05,
1196
+ "loss": 0.2618,
1197
+ "step": 396
1198
+ },
1199
+ {
1200
+ "epoch": 1.23,
1201
+ "learning_rate": 2e-05,
1202
+ "loss": 0.2647,
1203
+ "step": 398
1204
+ },
1205
+ {
1206
+ "epoch": 1.24,
1207
+ "learning_rate": 2e-05,
1208
+ "loss": 0.2592,
1209
+ "step": 400
1210
+ },
1211
+ {
1212
+ "epoch": 1.25,
1213
+ "learning_rate": 2e-05,
1214
+ "loss": 0.2633,
1215
+ "step": 402
1216
+ },
1217
+ {
1218
+ "epoch": 1.25,
1219
+ "learning_rate": 2e-05,
1220
+ "loss": 0.2651,
1221
+ "step": 404
1222
+ },
1223
+ {
1224
+ "epoch": 1.26,
1225
+ "learning_rate": 2e-05,
1226
+ "loss": 0.2573,
1227
+ "step": 406
1228
+ },
1229
+ {
1230
+ "epoch": 1.26,
1231
+ "learning_rate": 2e-05,
1232
+ "loss": 0.2633,
1233
+ "step": 408
1234
+ },
1235
+ {
1236
+ "epoch": 1.27,
1237
+ "learning_rate": 2e-05,
1238
+ "loss": 0.2537,
1239
+ "step": 410
1240
+ },
1241
+ {
1242
+ "epoch": 1.28,
1243
+ "learning_rate": 2e-05,
1244
+ "loss": 0.2625,
1245
+ "step": 412
1246
+ },
1247
+ {
1248
+ "epoch": 1.28,
1249
+ "learning_rate": 2e-05,
1250
+ "loss": 0.2636,
1251
+ "step": 414
1252
+ },
1253
+ {
1254
+ "epoch": 1.29,
1255
+ "learning_rate": 2e-05,
1256
+ "loss": 0.2549,
1257
+ "step": 416
1258
+ },
1259
+ {
1260
+ "epoch": 1.29,
1261
+ "learning_rate": 2e-05,
1262
+ "loss": 0.2596,
1263
+ "step": 418
1264
+ },
1265
+ {
1266
+ "epoch": 1.3,
1267
+ "learning_rate": 2e-05,
1268
+ "loss": 0.257,
1269
+ "step": 420
1270
+ },
1271
+ {
1272
+ "epoch": 1.31,
1273
+ "learning_rate": 2e-05,
1274
+ "loss": 0.2605,
1275
+ "step": 422
1276
+ },
1277
+ {
1278
+ "epoch": 1.31,
1279
+ "learning_rate": 2e-05,
1280
+ "loss": 0.2531,
1281
+ "step": 424
1282
+ },
1283
+ {
1284
+ "epoch": 1.32,
1285
+ "learning_rate": 2e-05,
1286
+ "loss": 0.2547,
1287
+ "step": 426
1288
+ },
1289
+ {
1290
+ "epoch": 1.33,
1291
+ "learning_rate": 2e-05,
1292
+ "loss": 0.255,
1293
+ "step": 428
1294
+ },
1295
+ {
1296
+ "epoch": 1.33,
1297
+ "learning_rate": 2e-05,
1298
+ "loss": 0.2518,
1299
+ "step": 430
1300
+ },
1301
+ {
1302
+ "epoch": 1.34,
1303
+ "learning_rate": 2e-05,
1304
+ "loss": 0.25,
1305
+ "step": 432
1306
+ },
1307
+ {
1308
+ "epoch": 1.34,
1309
+ "learning_rate": 2e-05,
1310
+ "loss": 0.2552,
1311
+ "step": 434
1312
+ },
1313
+ {
1314
+ "epoch": 1.35,
1315
+ "learning_rate": 2e-05,
1316
+ "loss": 0.2532,
1317
+ "step": 436
1318
+ },
1319
+ {
1320
+ "epoch": 1.36,
1321
+ "learning_rate": 2e-05,
1322
+ "loss": 0.2533,
1323
+ "step": 438
1324
+ },
1325
+ {
1326
+ "epoch": 1.36,
1327
+ "learning_rate": 2e-05,
1328
+ "loss": 0.2543,
1329
+ "step": 440
1330
+ },
1331
+ {
1332
+ "epoch": 1.37,
1333
+ "learning_rate": 2e-05,
1334
+ "loss": 0.2508,
1335
+ "step": 442
1336
+ },
1337
+ {
1338
+ "epoch": 1.38,
1339
+ "learning_rate": 2e-05,
1340
+ "loss": 0.2504,
1341
+ "step": 444
1342
+ },
1343
+ {
1344
+ "epoch": 1.38,
1345
+ "learning_rate": 2e-05,
1346
+ "loss": 0.2477,
1347
+ "step": 446
1348
+ },
1349
+ {
1350
+ "epoch": 1.39,
1351
+ "learning_rate": 2e-05,
1352
+ "loss": 0.2535,
1353
+ "step": 448
1354
+ },
1355
+ {
1356
+ "epoch": 1.39,
1357
+ "learning_rate": 2e-05,
1358
+ "loss": 0.2458,
1359
+ "step": 450
1360
+ },
1361
+ {
1362
+ "epoch": 1.4,
1363
+ "learning_rate": 2e-05,
1364
+ "loss": 0.2494,
1365
+ "step": 452
1366
+ },
1367
+ {
1368
+ "epoch": 1.41,
1369
+ "learning_rate": 2e-05,
1370
+ "loss": 0.2466,
1371
+ "step": 454
1372
+ },
1373
+ {
1374
+ "epoch": 1.41,
1375
+ "learning_rate": 2e-05,
1376
+ "loss": 0.2472,
1377
+ "step": 456
1378
+ },
1379
+ {
1380
+ "epoch": 1.42,
1381
+ "learning_rate": 2e-05,
1382
+ "loss": 0.2477,
1383
+ "step": 458
1384
+ },
1385
+ {
1386
+ "epoch": 1.42,
1387
+ "learning_rate": 2e-05,
1388
+ "loss": 0.2501,
1389
+ "step": 460
1390
+ },
1391
+ {
1392
+ "epoch": 1.43,
1393
+ "learning_rate": 2e-05,
1394
+ "loss": 0.2534,
1395
+ "step": 462
1396
+ },
1397
+ {
1398
+ "epoch": 1.44,
1399
+ "learning_rate": 2e-05,
1400
+ "loss": 0.2443,
1401
+ "step": 464
1402
+ },
1403
+ {
1404
+ "epoch": 1.44,
1405
+ "learning_rate": 2e-05,
1406
+ "loss": 0.2463,
1407
+ "step": 466
1408
+ },
1409
+ {
1410
+ "epoch": 1.45,
1411
+ "learning_rate": 2e-05,
1412
+ "loss": 0.2426,
1413
+ "step": 468
1414
+ },
1415
+ {
1416
+ "epoch": 1.46,
1417
+ "learning_rate": 2e-05,
1418
+ "loss": 0.246,
1419
+ "step": 470
1420
+ },
1421
+ {
1422
+ "epoch": 1.46,
1423
+ "learning_rate": 2e-05,
1424
+ "loss": 0.2474,
1425
+ "step": 472
1426
+ },
1427
+ {
1428
+ "epoch": 1.47,
1429
+ "learning_rate": 2e-05,
1430
+ "loss": 0.247,
1431
+ "step": 474
1432
+ },
1433
+ {
1434
+ "epoch": 1.47,
1435
+ "learning_rate": 2e-05,
1436
+ "loss": 0.241,
1437
+ "step": 476
1438
+ },
1439
+ {
1440
+ "epoch": 1.48,
1441
+ "learning_rate": 2e-05,
1442
+ "loss": 0.2498,
1443
+ "step": 478
1444
+ },
1445
+ {
1446
+ "epoch": 1.49,
1447
+ "learning_rate": 2e-05,
1448
+ "loss": 0.2443,
1449
+ "step": 480
1450
+ },
1451
+ {
1452
+ "epoch": 1.49,
1453
+ "learning_rate": 2e-05,
1454
+ "loss": 0.2503,
1455
+ "step": 482
1456
+ },
1457
+ {
1458
+ "epoch": 1.5,
1459
+ "learning_rate": 2e-05,
1460
+ "loss": 0.2483,
1461
+ "step": 484
1462
+ },
1463
+ {
1464
+ "epoch": 1.51,
1465
+ "learning_rate": 2e-05,
1466
+ "loss": 0.2467,
1467
+ "step": 486
1468
+ },
1469
+ {
1470
+ "epoch": 1.51,
1471
+ "learning_rate": 2e-05,
1472
+ "loss": 0.2473,
1473
+ "step": 488
1474
+ },
1475
+ {
1476
+ "epoch": 1.52,
1477
+ "learning_rate": 2e-05,
1478
+ "loss": 0.2412,
1479
+ "step": 490
1480
+ },
1481
+ {
1482
+ "epoch": 1.52,
1483
+ "learning_rate": 2e-05,
1484
+ "loss": 0.2401,
1485
+ "step": 492
1486
+ },
1487
+ {
1488
+ "epoch": 1.53,
1489
+ "learning_rate": 2e-05,
1490
+ "loss": 0.2448,
1491
+ "step": 494
1492
+ },
1493
+ {
1494
+ "epoch": 1.54,
1495
+ "learning_rate": 2e-05,
1496
+ "loss": 0.2373,
1497
+ "step": 496
1498
+ },
1499
+ {
1500
+ "epoch": 1.54,
1501
+ "learning_rate": 2e-05,
1502
+ "loss": 0.2425,
1503
+ "step": 498
1504
+ },
1505
+ {
1506
+ "epoch": 1.55,
1507
+ "learning_rate": 2e-05,
1508
+ "loss": 0.2375,
1509
+ "step": 500
1510
+ },
1511
+ {
1512
+ "epoch": 1.55,
1513
+ "learning_rate": 2e-05,
1514
+ "loss": 0.2441,
1515
+ "step": 502
1516
+ },
1517
+ {
1518
+ "epoch": 1.56,
1519
+ "learning_rate": 2e-05,
1520
+ "loss": 0.2383,
1521
+ "step": 504
1522
+ },
1523
+ {
1524
+ "epoch": 1.57,
1525
+ "learning_rate": 2e-05,
1526
+ "loss": 0.2471,
1527
+ "step": 506
1528
+ },
1529
+ {
1530
+ "epoch": 1.57,
1531
+ "learning_rate": 2e-05,
1532
+ "loss": 0.2385,
1533
+ "step": 508
1534
+ },
1535
+ {
1536
+ "epoch": 1.58,
1537
+ "learning_rate": 2e-05,
1538
+ "loss": 0.2385,
1539
+ "step": 510
1540
+ },
1541
+ {
1542
+ "epoch": 1.59,
1543
+ "learning_rate": 2e-05,
1544
+ "loss": 0.2411,
1545
+ "step": 512
1546
+ },
1547
+ {
1548
+ "epoch": 1.59,
1549
+ "learning_rate": 2e-05,
1550
+ "loss": 0.2346,
1551
+ "step": 514
1552
+ },
1553
+ {
1554
+ "epoch": 1.6,
1555
+ "learning_rate": 2e-05,
1556
+ "loss": 0.2373,
1557
+ "step": 516
1558
+ },
1559
+ {
1560
+ "epoch": 1.6,
1561
+ "learning_rate": 2e-05,
1562
+ "loss": 0.2404,
1563
+ "step": 518
1564
+ },
1565
+ {
1566
+ "epoch": 1.61,
1567
+ "learning_rate": 2e-05,
1568
+ "loss": 0.2403,
1569
+ "step": 520
1570
+ },
1571
+ {
1572
+ "epoch": 1.62,
1573
+ "learning_rate": 2e-05,
1574
+ "loss": 0.2413,
1575
+ "step": 522
1576
+ },
1577
+ {
1578
+ "epoch": 1.62,
1579
+ "learning_rate": 2e-05,
1580
+ "loss": 0.2323,
1581
+ "step": 524
1582
+ },
1583
+ {
1584
+ "epoch": 1.63,
1585
+ "learning_rate": 2e-05,
1586
+ "loss": 0.2376,
1587
+ "step": 526
1588
+ },
1589
+ {
1590
+ "epoch": 1.64,
1591
+ "learning_rate": 2e-05,
1592
+ "loss": 0.2362,
1593
+ "step": 528
1594
+ },
1595
+ {
1596
+ "epoch": 1.64,
1597
+ "learning_rate": 2e-05,
1598
+ "loss": 0.2359,
1599
+ "step": 530
1600
+ },
1601
+ {
1602
+ "epoch": 1.65,
1603
+ "learning_rate": 2e-05,
1604
+ "loss": 0.2376,
1605
+ "step": 532
1606
+ },
1607
+ {
1608
+ "epoch": 1.65,
1609
+ "learning_rate": 2e-05,
1610
+ "loss": 0.2351,
1611
+ "step": 534
1612
+ },
1613
+ {
1614
+ "epoch": 1.66,
1615
+ "learning_rate": 2e-05,
1616
+ "loss": 0.2341,
1617
+ "step": 536
1618
+ },
1619
+ {
1620
+ "epoch": 1.67,
1621
+ "learning_rate": 2e-05,
1622
+ "loss": 0.2345,
1623
+ "step": 538
1624
+ },
1625
+ {
1626
+ "epoch": 1.67,
1627
+ "learning_rate": 2e-05,
1628
+ "loss": 0.2354,
1629
+ "step": 540
1630
+ },
1631
+ {
1632
+ "epoch": 1.68,
1633
+ "learning_rate": 2e-05,
1634
+ "loss": 0.2365,
1635
+ "step": 542
1636
+ },
1637
+ {
1638
+ "epoch": 1.68,
1639
+ "learning_rate": 2e-05,
1640
+ "loss": 0.2335,
1641
+ "step": 544
1642
+ },
1643
+ {
1644
+ "epoch": 1.69,
1645
+ "learning_rate": 2e-05,
1646
+ "loss": 0.2342,
1647
+ "step": 546
1648
+ },
1649
+ {
1650
+ "epoch": 1.7,
1651
+ "learning_rate": 2e-05,
1652
+ "loss": 0.2319,
1653
+ "step": 548
1654
+ },
1655
+ {
1656
+ "epoch": 1.7,
1657
+ "learning_rate": 2e-05,
1658
+ "loss": 0.2388,
1659
+ "step": 550
1660
+ },
1661
+ {
1662
+ "epoch": 1.71,
1663
+ "learning_rate": 2e-05,
1664
+ "loss": 0.2362,
1665
+ "step": 552
1666
+ },
1667
+ {
1668
+ "epoch": 1.72,
1669
+ "learning_rate": 2e-05,
1670
+ "loss": 0.2342,
1671
+ "step": 554
1672
+ },
1673
+ {
1674
+ "epoch": 1.72,
1675
+ "learning_rate": 2e-05,
1676
+ "loss": 0.2282,
1677
+ "step": 556
1678
+ },
1679
+ {
1680
+ "epoch": 1.73,
1681
+ "learning_rate": 2e-05,
1682
+ "loss": 0.2354,
1683
+ "step": 558
1684
+ },
1685
+ {
1686
+ "epoch": 1.73,
1687
+ "learning_rate": 2e-05,
1688
+ "loss": 0.2337,
1689
+ "step": 560
1690
+ },
1691
+ {
1692
+ "epoch": 1.74,
1693
+ "learning_rate": 2e-05,
1694
+ "loss": 0.2286,
1695
+ "step": 562
1696
+ },
1697
+ {
1698
+ "epoch": 1.75,
1699
+ "learning_rate": 2e-05,
1700
+ "loss": 0.2323,
1701
+ "step": 564
1702
+ },
1703
+ {
1704
+ "epoch": 1.75,
1705
+ "learning_rate": 2e-05,
1706
+ "loss": 0.2298,
1707
+ "step": 566
1708
+ },
1709
+ {
1710
+ "epoch": 1.76,
1711
+ "learning_rate": 2e-05,
1712
+ "loss": 0.2302,
1713
+ "step": 568
1714
+ },
1715
+ {
1716
+ "epoch": 1.77,
1717
+ "learning_rate": 2e-05,
1718
+ "loss": 0.2296,
1719
+ "step": 570
1720
+ },
1721
+ {
1722
+ "epoch": 1.77,
1723
+ "learning_rate": 2e-05,
1724
+ "loss": 0.2329,
1725
+ "step": 572
1726
+ },
1727
+ {
1728
+ "epoch": 1.78,
1729
+ "learning_rate": 2e-05,
1730
+ "loss": 0.2298,
1731
+ "step": 574
1732
+ },
1733
+ {
1734
+ "epoch": 1.78,
1735
+ "learning_rate": 2e-05,
1736
+ "loss": 0.228,
1737
+ "step": 576
1738
+ },
1739
+ {
1740
+ "epoch": 1.79,
1741
+ "learning_rate": 2e-05,
1742
+ "loss": 0.2262,
1743
+ "step": 578
1744
+ },
1745
+ {
1746
+ "epoch": 1.8,
1747
+ "learning_rate": 2e-05,
1748
+ "loss": 0.2296,
1749
+ "step": 580
1750
+ },
1751
+ {
1752
+ "epoch": 1.8,
1753
+ "learning_rate": 2e-05,
1754
+ "loss": 0.2284,
1755
+ "step": 582
1756
+ },
1757
+ {
1758
+ "epoch": 1.81,
1759
+ "learning_rate": 2e-05,
1760
+ "loss": 0.2299,
1761
+ "step": 584
1762
+ },
1763
+ {
1764
+ "epoch": 1.81,
1765
+ "learning_rate": 2e-05,
1766
+ "loss": 0.2276,
1767
+ "step": 586
1768
+ },
1769
+ {
1770
+ "epoch": 1.82,
1771
+ "learning_rate": 2e-05,
1772
+ "loss": 0.2315,
1773
+ "step": 588
1774
+ },
1775
+ {
1776
+ "epoch": 1.83,
1777
+ "learning_rate": 2e-05,
1778
+ "loss": 0.2304,
1779
+ "step": 590
1780
+ },
1781
+ {
1782
+ "epoch": 1.83,
1783
+ "learning_rate": 2e-05,
1784
+ "loss": 0.2253,
1785
+ "step": 592
1786
+ },
1787
+ {
1788
+ "epoch": 1.84,
1789
+ "learning_rate": 2e-05,
1790
+ "loss": 0.2243,
1791
+ "step": 594
1792
+ },
1793
+ {
1794
+ "epoch": 1.85,
1795
+ "learning_rate": 2e-05,
1796
+ "loss": 0.2282,
1797
+ "step": 596
1798
+ },
1799
+ {
1800
+ "epoch": 1.85,
1801
+ "learning_rate": 2e-05,
1802
+ "loss": 0.2277,
1803
+ "step": 598
1804
+ },
1805
+ {
1806
+ "epoch": 1.86,
1807
+ "learning_rate": 2e-05,
1808
+ "loss": 0.2306,
1809
+ "step": 600
1810
+ },
1811
+ {
1812
+ "epoch": 1.86,
1813
+ "learning_rate": 2e-05,
1814
+ "loss": 0.23,
1815
+ "step": 602
1816
+ },
1817
+ {
1818
+ "epoch": 1.87,
1819
+ "learning_rate": 2e-05,
1820
+ "loss": 0.2295,
1821
+ "step": 604
1822
+ },
1823
+ {
1824
+ "epoch": 1.88,
1825
+ "learning_rate": 2e-05,
1826
+ "loss": 0.2266,
1827
+ "step": 606
1828
+ },
1829
+ {
1830
+ "epoch": 1.88,
1831
+ "learning_rate": 2e-05,
1832
+ "loss": 0.2191,
1833
+ "step": 608
1834
+ },
1835
+ {
1836
+ "epoch": 1.89,
1837
+ "learning_rate": 2e-05,
1838
+ "loss": 0.2219,
1839
+ "step": 610
1840
+ },
1841
+ {
1842
+ "epoch": 1.9,
1843
+ "learning_rate": 2e-05,
1844
+ "loss": 0.2277,
1845
+ "step": 612
1846
+ },
1847
+ {
1848
+ "epoch": 1.9,
1849
+ "learning_rate": 2e-05,
1850
+ "loss": 0.2237,
1851
+ "step": 614
1852
+ },
1853
+ {
1854
+ "epoch": 1.91,
1855
+ "learning_rate": 2e-05,
1856
+ "loss": 0.2229,
1857
+ "step": 616
1858
+ },
1859
+ {
1860
+ "epoch": 1.91,
1861
+ "learning_rate": 2e-05,
1862
+ "loss": 0.2209,
1863
+ "step": 618
1864
+ },
1865
+ {
1866
+ "epoch": 1.92,
1867
+ "learning_rate": 2e-05,
1868
+ "loss": 0.2229,
1869
+ "step": 620
1870
+ },
1871
+ {
1872
+ "epoch": 1.93,
1873
+ "learning_rate": 2e-05,
1874
+ "loss": 0.2208,
1875
+ "step": 622
1876
+ },
1877
+ {
1878
+ "epoch": 1.93,
1879
+ "learning_rate": 2e-05,
1880
+ "loss": 0.2245,
1881
+ "step": 624
1882
+ },
1883
+ {
1884
+ "epoch": 1.94,
1885
+ "learning_rate": 2e-05,
1886
+ "loss": 0.2245,
1887
+ "step": 626
1888
+ },
1889
+ {
1890
+ "epoch": 1.95,
1891
+ "learning_rate": 2e-05,
1892
+ "loss": 0.2219,
1893
+ "step": 628
1894
+ },
1895
+ {
1896
+ "epoch": 1.95,
1897
+ "learning_rate": 2e-05,
1898
+ "loss": 0.2218,
1899
+ "step": 630
1900
+ },
1901
+ {
1902
+ "epoch": 1.96,
1903
+ "learning_rate": 2e-05,
1904
+ "loss": 0.218,
1905
+ "step": 632
1906
+ },
1907
+ {
1908
+ "epoch": 1.96,
1909
+ "learning_rate": 2e-05,
1910
+ "loss": 0.2238,
1911
+ "step": 634
1912
+ },
1913
+ {
1914
+ "epoch": 1.97,
1915
+ "learning_rate": 2e-05,
1916
+ "loss": 0.2176,
1917
+ "step": 636
1918
+ },
1919
+ {
1920
+ "epoch": 1.98,
1921
+ "learning_rate": 2e-05,
1922
+ "loss": 0.2167,
1923
+ "step": 638
1924
+ },
1925
+ {
1926
+ "epoch": 1.98,
1927
+ "learning_rate": 2e-05,
1928
+ "loss": 0.2226,
1929
+ "step": 640
1930
+ },
1931
+ {
1932
+ "epoch": 1.99,
1933
+ "learning_rate": 2e-05,
1934
+ "loss": 0.2182,
1935
+ "step": 642
1936
+ },
1937
+ {
1938
+ "epoch": 1.99,
1939
+ "learning_rate": 2e-05,
1940
+ "loss": 0.2144,
1941
+ "step": 644
1942
+ },
1943
+ {
1944
+ "epoch": 2.0,
1945
+ "learning_rate": 2e-05,
1946
+ "loss": 0.2122,
1947
+ "step": 646
1948
+ },
1949
+ {
1950
+ "epoch": 2.01,
1951
+ "learning_rate": 2e-05,
1952
+ "loss": 0.201,
1953
+ "step": 648
1954
+ },
1955
+ {
1956
+ "epoch": 2.01,
1957
+ "learning_rate": 2e-05,
1958
+ "loss": 0.2033,
1959
+ "step": 650
1960
+ },
1961
+ {
1962
+ "epoch": 2.02,
1963
+ "learning_rate": 2e-05,
1964
+ "loss": 0.1941,
1965
+ "step": 652
1966
+ },
1967
+ {
1968
+ "epoch": 2.03,
1969
+ "learning_rate": 2e-05,
1970
+ "loss": 0.2006,
1971
+ "step": 654
1972
+ },
1973
+ {
1974
+ "epoch": 2.03,
1975
+ "learning_rate": 2e-05,
1976
+ "loss": 0.1975,
1977
+ "step": 656
1978
+ },
1979
+ {
1980
+ "epoch": 2.04,
1981
+ "learning_rate": 2e-05,
1982
+ "loss": 0.2006,
1983
+ "step": 658
1984
+ },
1985
+ {
1986
+ "epoch": 2.04,
1987
+ "learning_rate": 2e-05,
1988
+ "loss": 0.1975,
1989
+ "step": 660
1990
+ },
1991
+ {
1992
+ "epoch": 2.05,
1993
+ "learning_rate": 2e-05,
1994
+ "loss": 0.1962,
1995
+ "step": 662
1996
+ },
1997
+ {
1998
+ "epoch": 2.06,
1999
+ "learning_rate": 2e-05,
2000
+ "loss": 0.195,
2001
+ "step": 664
2002
+ },
2003
+ {
2004
+ "epoch": 2.06,
2005
+ "learning_rate": 2e-05,
2006
+ "loss": 0.1967,
2007
+ "step": 666
2008
+ },
2009
+ {
2010
+ "epoch": 2.07,
2011
+ "learning_rate": 2e-05,
2012
+ "loss": 0.1918,
2013
+ "step": 668
2014
+ },
2015
+ {
2016
+ "epoch": 2.08,
2017
+ "learning_rate": 2e-05,
2018
+ "loss": 0.1927,
2019
+ "step": 670
2020
+ },
2021
+ {
2022
+ "epoch": 2.08,
2023
+ "learning_rate": 2e-05,
2024
+ "loss": 0.1934,
2025
+ "step": 672
2026
+ },
2027
+ {
2028
+ "epoch": 2.09,
2029
+ "learning_rate": 2e-05,
2030
+ "loss": 0.1963,
2031
+ "step": 674
2032
+ },
2033
+ {
2034
+ "epoch": 2.09,
2035
+ "learning_rate": 2e-05,
2036
+ "loss": 0.1951,
2037
+ "step": 676
2038
+ },
2039
+ {
2040
+ "epoch": 2.1,
2041
+ "learning_rate": 2e-05,
2042
+ "loss": 0.1886,
2043
+ "step": 678
2044
+ },
2045
+ {
2046
+ "epoch": 2.11,
2047
+ "learning_rate": 2e-05,
2048
+ "loss": 0.1939,
2049
+ "step": 680
2050
+ },
2051
+ {
2052
+ "epoch": 2.11,
2053
+ "learning_rate": 2e-05,
2054
+ "loss": 0.1978,
2055
+ "step": 682
2056
+ },
2057
+ {
2058
+ "epoch": 2.12,
2059
+ "learning_rate": 2e-05,
2060
+ "loss": 0.1941,
2061
+ "step": 684
2062
+ },
2063
+ {
2064
+ "epoch": 2.12,
2065
+ "learning_rate": 2e-05,
2066
+ "loss": 0.1955,
2067
+ "step": 686
2068
+ },
2069
+ {
2070
+ "epoch": 2.13,
2071
+ "learning_rate": 2e-05,
2072
+ "loss": 0.194,
2073
+ "step": 688
2074
+ },
2075
+ {
2076
+ "epoch": 2.14,
2077
+ "learning_rate": 2e-05,
2078
+ "loss": 0.1971,
2079
+ "step": 690
2080
+ },
2081
+ {
2082
+ "epoch": 2.14,
2083
+ "learning_rate": 2e-05,
2084
+ "loss": 0.19,
2085
+ "step": 692
2086
+ },
2087
+ {
2088
+ "epoch": 2.15,
2089
+ "learning_rate": 2e-05,
2090
+ "loss": 0.1887,
2091
+ "step": 694
2092
+ },
2093
+ {
2094
+ "epoch": 2.16,
2095
+ "learning_rate": 2e-05,
2096
+ "loss": 0.1923,
2097
+ "step": 696
2098
+ },
2099
+ {
2100
+ "epoch": 2.16,
2101
+ "learning_rate": 2e-05,
2102
+ "loss": 0.1909,
2103
+ "step": 698
2104
+ },
2105
+ {
2106
+ "epoch": 2.17,
2107
+ "learning_rate": 2e-05,
2108
+ "loss": 0.1932,
2109
+ "step": 700
2110
+ },
2111
+ {
2112
+ "epoch": 2.17,
2113
+ "learning_rate": 2e-05,
2114
+ "loss": 0.1929,
2115
+ "step": 702
2116
+ },
2117
+ {
2118
+ "epoch": 2.18,
2119
+ "learning_rate": 2e-05,
2120
+ "loss": 0.1981,
2121
+ "step": 704
2122
+ },
2123
+ {
2124
+ "epoch": 2.19,
2125
+ "learning_rate": 2e-05,
2126
+ "loss": 0.1923,
2127
+ "step": 706
2128
+ },
2129
+ {
2130
+ "epoch": 2.19,
2131
+ "learning_rate": 2e-05,
2132
+ "loss": 0.1896,
2133
+ "step": 708
2134
+ },
2135
+ {
2136
+ "epoch": 2.2,
2137
+ "learning_rate": 2e-05,
2138
+ "loss": 0.1946,
2139
+ "step": 710
2140
+ },
2141
+ {
2142
+ "epoch": 2.21,
2143
+ "learning_rate": 2e-05,
2144
+ "loss": 0.1913,
2145
+ "step": 712
2146
+ },
2147
+ {
2148
+ "epoch": 2.21,
2149
+ "learning_rate": 2e-05,
2150
+ "loss": 0.1938,
2151
+ "step": 714
2152
+ },
2153
+ {
2154
+ "epoch": 2.22,
2155
+ "learning_rate": 2e-05,
2156
+ "loss": 0.1949,
2157
+ "step": 716
2158
+ },
2159
+ {
2160
+ "epoch": 2.22,
2161
+ "learning_rate": 2e-05,
2162
+ "loss": 0.1935,
2163
+ "step": 718
2164
+ },
2165
+ {
2166
+ "epoch": 2.23,
2167
+ "learning_rate": 2e-05,
2168
+ "loss": 0.1912,
2169
+ "step": 720
2170
+ },
2171
+ {
2172
+ "epoch": 2.24,
2173
+ "learning_rate": 2e-05,
2174
+ "loss": 0.1934,
2175
+ "step": 722
2176
+ },
2177
+ {
2178
+ "epoch": 2.24,
2179
+ "learning_rate": 2e-05,
2180
+ "loss": 0.1953,
2181
+ "step": 724
2182
+ },
2183
+ {
2184
+ "epoch": 2.25,
2185
+ "learning_rate": 2e-05,
2186
+ "loss": 0.1892,
2187
+ "step": 726
2188
+ },
2189
+ {
2190
+ "epoch": 2.25,
2191
+ "learning_rate": 2e-05,
2192
+ "loss": 0.1918,
2193
+ "step": 728
2194
+ },
2195
+ {
2196
+ "epoch": 2.26,
2197
+ "learning_rate": 2e-05,
2198
+ "loss": 0.1923,
2199
+ "step": 730
2200
+ },
2201
+ {
2202
+ "epoch": 2.27,
2203
+ "learning_rate": 2e-05,
2204
+ "loss": 0.1818,
2205
+ "step": 732
2206
+ },
2207
+ {
2208
+ "epoch": 2.27,
2209
+ "learning_rate": 2e-05,
2210
+ "loss": 0.1873,
2211
+ "step": 734
2212
+ },
2213
+ {
2214
+ "epoch": 2.28,
2215
+ "learning_rate": 2e-05,
2216
+ "loss": 0.1898,
2217
+ "step": 736
2218
+ },
2219
+ {
2220
+ "epoch": 2.29,
2221
+ "learning_rate": 2e-05,
2222
+ "loss": 0.1882,
2223
+ "step": 738
2224
+ },
2225
+ {
2226
+ "epoch": 2.29,
2227
+ "learning_rate": 2e-05,
2228
+ "loss": 0.1854,
2229
+ "step": 740
2230
+ },
2231
+ {
2232
+ "epoch": 2.3,
2233
+ "learning_rate": 2e-05,
2234
+ "loss": 0.189,
2235
+ "step": 742
2236
+ },
2237
+ {
2238
+ "epoch": 2.3,
2239
+ "learning_rate": 2e-05,
2240
+ "loss": 0.1919,
2241
+ "step": 744
2242
+ },
2243
+ {
2244
+ "epoch": 2.31,
2245
+ "learning_rate": 2e-05,
2246
+ "loss": 0.1902,
2247
+ "step": 746
2248
+ },
2249
+ {
2250
+ "epoch": 2.32,
2251
+ "learning_rate": 2e-05,
2252
+ "loss": 0.1865,
2253
+ "step": 748
2254
+ },
2255
+ {
2256
+ "epoch": 2.32,
2257
+ "learning_rate": 2e-05,
2258
+ "loss": 0.1858,
2259
+ "step": 750
2260
+ },
2261
+ {
2262
+ "epoch": 2.33,
2263
+ "learning_rate": 2e-05,
2264
+ "loss": 0.1833,
2265
+ "step": 752
2266
+ },
2267
+ {
2268
+ "epoch": 2.34,
2269
+ "learning_rate": 2e-05,
2270
+ "loss": 0.1829,
2271
+ "step": 754
2272
+ },
2273
+ {
2274
+ "epoch": 2.34,
2275
+ "learning_rate": 2e-05,
2276
+ "loss": 0.1845,
2277
+ "step": 756
2278
+ },
2279
+ {
2280
+ "epoch": 2.35,
2281
+ "learning_rate": 2e-05,
2282
+ "loss": 0.182,
2283
+ "step": 758
2284
+ },
2285
+ {
2286
+ "epoch": 2.35,
2287
+ "learning_rate": 2e-05,
2288
+ "loss": 0.1854,
2289
+ "step": 760
2290
+ },
2291
+ {
2292
+ "epoch": 2.36,
2293
+ "learning_rate": 2e-05,
2294
+ "loss": 0.1876,
2295
+ "step": 762
2296
+ },
2297
+ {
2298
+ "epoch": 2.37,
2299
+ "learning_rate": 2e-05,
2300
+ "loss": 0.1868,
2301
+ "step": 764
2302
+ },
2303
+ {
2304
+ "epoch": 2.37,
2305
+ "learning_rate": 2e-05,
2306
+ "loss": 0.1911,
2307
+ "step": 766
2308
+ },
2309
+ {
2310
+ "epoch": 2.38,
2311
+ "learning_rate": 2e-05,
2312
+ "loss": 0.1804,
2313
+ "step": 768
2314
+ },
2315
+ {
2316
+ "epoch": 2.38,
2317
+ "learning_rate": 2e-05,
2318
+ "loss": 0.1791,
2319
+ "step": 770
2320
+ },
2321
+ {
2322
+ "epoch": 2.39,
2323
+ "learning_rate": 2e-05,
2324
+ "loss": 0.1889,
2325
+ "step": 772
2326
+ },
2327
+ {
2328
+ "epoch": 2.4,
2329
+ "learning_rate": 2e-05,
2330
+ "loss": 0.1827,
2331
+ "step": 774
2332
+ },
2333
+ {
2334
+ "epoch": 2.4,
2335
+ "learning_rate": 2e-05,
2336
+ "loss": 0.1854,
2337
+ "step": 776
2338
+ },
2339
+ {
2340
+ "epoch": 2.41,
2341
+ "learning_rate": 2e-05,
2342
+ "loss": 0.1826,
2343
+ "step": 778
2344
+ },
2345
+ {
2346
+ "epoch": 2.42,
2347
+ "learning_rate": 2e-05,
2348
+ "loss": 0.1793,
2349
+ "step": 780
2350
+ },
2351
+ {
2352
+ "epoch": 2.42,
2353
+ "learning_rate": 2e-05,
2354
+ "loss": 0.1859,
2355
+ "step": 782
2356
+ },
2357
+ {
2358
+ "epoch": 2.43,
2359
+ "learning_rate": 2e-05,
2360
+ "loss": 0.1827,
2361
+ "step": 784
2362
+ },
2363
+ {
2364
+ "epoch": 2.43,
2365
+ "learning_rate": 2e-05,
2366
+ "loss": 0.1894,
2367
+ "step": 786
2368
+ },
2369
+ {
2370
+ "epoch": 2.44,
2371
+ "learning_rate": 2e-05,
2372
+ "loss": 0.1878,
2373
+ "step": 788
2374
+ },
2375
+ {
2376
+ "epoch": 2.45,
2377
+ "learning_rate": 2e-05,
2378
+ "loss": 0.1858,
2379
+ "step": 790
2380
+ },
2381
+ {
2382
+ "epoch": 2.45,
2383
+ "learning_rate": 2e-05,
2384
+ "loss": 0.1836,
2385
+ "step": 792
2386
+ },
2387
+ {
2388
+ "epoch": 2.46,
2389
+ "learning_rate": 2e-05,
2390
+ "loss": 0.184,
2391
+ "step": 794
2392
+ },
2393
+ {
2394
+ "epoch": 2.47,
2395
+ "learning_rate": 2e-05,
2396
+ "loss": 0.1828,
2397
+ "step": 796
2398
+ },
2399
+ {
2400
+ "epoch": 2.47,
2401
+ "learning_rate": 2e-05,
2402
+ "loss": 0.1804,
2403
+ "step": 798
2404
+ },
2405
+ {
2406
+ "epoch": 2.48,
2407
+ "learning_rate": 2e-05,
2408
+ "loss": 0.1764,
2409
+ "step": 800
2410
+ },
2411
+ {
2412
+ "epoch": 2.48,
2413
+ "learning_rate": 2e-05,
2414
+ "loss": 0.181,
2415
+ "step": 802
2416
+ },
2417
+ {
2418
+ "epoch": 2.49,
2419
+ "learning_rate": 2e-05,
2420
+ "loss": 0.1881,
2421
+ "step": 804
2422
+ },
2423
+ {
2424
+ "epoch": 2.5,
2425
+ "learning_rate": 2e-05,
2426
+ "loss": 0.1848,
2427
+ "step": 806
2428
+ },
2429
+ {
2430
+ "epoch": 2.5,
2431
+ "learning_rate": 2e-05,
2432
+ "loss": 0.1799,
2433
+ "step": 808
2434
+ },
2435
+ {
2436
+ "epoch": 2.51,
2437
+ "learning_rate": 2e-05,
2438
+ "loss": 0.1846,
2439
+ "step": 810
2440
+ },
2441
+ {
2442
+ "epoch": 2.51,
2443
+ "learning_rate": 2e-05,
2444
+ "loss": 0.1799,
2445
+ "step": 812
2446
+ },
2447
+ {
2448
+ "epoch": 2.52,
2449
+ "learning_rate": 2e-05,
2450
+ "loss": 0.1805,
2451
+ "step": 814
2452
+ },
2453
+ {
2454
+ "epoch": 2.53,
2455
+ "learning_rate": 2e-05,
2456
+ "loss": 0.1825,
2457
+ "step": 816
2458
+ },
2459
+ {
2460
+ "epoch": 2.53,
2461
+ "learning_rate": 2e-05,
2462
+ "loss": 0.1804,
2463
+ "step": 818
2464
+ },
2465
+ {
2466
+ "epoch": 2.54,
2467
+ "learning_rate": 2e-05,
2468
+ "loss": 0.1805,
2469
+ "step": 820
2470
+ },
2471
+ {
2472
+ "epoch": 2.55,
2473
+ "learning_rate": 2e-05,
2474
+ "loss": 0.1799,
2475
+ "step": 822
2476
+ },
2477
+ {
2478
+ "epoch": 2.55,
2479
+ "learning_rate": 2e-05,
2480
+ "loss": 0.1797,
2481
+ "step": 824
2482
+ },
2483
+ {
2484
+ "epoch": 2.56,
2485
+ "learning_rate": 2e-05,
2486
+ "loss": 0.1778,
2487
+ "step": 826
2488
+ },
2489
+ {
2490
+ "epoch": 2.56,
2491
+ "learning_rate": 2e-05,
2492
+ "loss": 0.1851,
2493
+ "step": 828
2494
+ },
2495
+ {
2496
+ "epoch": 2.57,
2497
+ "learning_rate": 2e-05,
2498
+ "loss": 0.1743,
2499
+ "step": 830
2500
+ },
2501
+ {
2502
+ "epoch": 2.58,
2503
+ "learning_rate": 2e-05,
2504
+ "loss": 0.1766,
2505
+ "step": 832
2506
+ },
2507
+ {
2508
+ "epoch": 2.58,
2509
+ "learning_rate": 2e-05,
2510
+ "loss": 0.1852,
2511
+ "step": 834
2512
+ },
2513
+ {
2514
+ "epoch": 2.59,
2515
+ "learning_rate": 2e-05,
2516
+ "loss": 0.1737,
2517
+ "step": 836
2518
+ },
2519
+ {
2520
+ "epoch": 2.6,
2521
+ "learning_rate": 2e-05,
2522
+ "loss": 0.1758,
2523
+ "step": 838
2524
+ },
2525
+ {
2526
+ "epoch": 2.6,
2527
+ "learning_rate": 2e-05,
2528
+ "loss": 0.1754,
2529
+ "step": 840
2530
+ },
2531
+ {
2532
+ "epoch": 2.61,
2533
+ "learning_rate": 2e-05,
2534
+ "loss": 0.1754,
2535
+ "step": 842
2536
+ },
2537
+ {
2538
+ "epoch": 2.61,
2539
+ "learning_rate": 2e-05,
2540
+ "loss": 0.1757,
2541
+ "step": 844
2542
+ },
2543
+ {
2544
+ "epoch": 2.62,
2545
+ "learning_rate": 2e-05,
2546
+ "loss": 0.1755,
2547
+ "step": 846
2548
+ },
2549
+ {
2550
+ "epoch": 2.63,
2551
+ "learning_rate": 2e-05,
2552
+ "loss": 0.1755,
2553
+ "step": 848
2554
+ },
2555
+ {
2556
+ "epoch": 2.63,
2557
+ "learning_rate": 2e-05,
2558
+ "loss": 0.1754,
2559
+ "step": 850
2560
+ },
2561
+ {
2562
+ "epoch": 2.64,
2563
+ "learning_rate": 2e-05,
2564
+ "loss": 0.1751,
2565
+ "step": 852
2566
+ },
2567
+ {
2568
+ "epoch": 2.64,
2569
+ "learning_rate": 2e-05,
2570
+ "loss": 0.1726,
2571
+ "step": 854
2572
+ },
2573
+ {
2574
+ "epoch": 2.65,
2575
+ "learning_rate": 2e-05,
2576
+ "loss": 0.1727,
2577
+ "step": 856
2578
+ },
2579
+ {
2580
+ "epoch": 2.66,
2581
+ "learning_rate": 2e-05,
2582
+ "loss": 0.1743,
2583
+ "step": 858
2584
+ },
2585
+ {
2586
+ "epoch": 2.66,
2587
+ "learning_rate": 2e-05,
2588
+ "loss": 0.1769,
2589
+ "step": 860
2590
+ },
2591
+ {
2592
+ "epoch": 2.67,
2593
+ "learning_rate": 2e-05,
2594
+ "loss": 0.1833,
2595
+ "step": 862
2596
+ },
2597
+ {
2598
+ "epoch": 2.68,
2599
+ "learning_rate": 2e-05,
2600
+ "loss": 0.1737,
2601
+ "step": 864
2602
+ },
2603
+ {
2604
+ "epoch": 2.68,
2605
+ "learning_rate": 2e-05,
2606
+ "loss": 0.1762,
2607
+ "step": 866
2608
+ },
2609
+ {
2610
+ "epoch": 2.69,
2611
+ "learning_rate": 2e-05,
2612
+ "loss": 0.172,
2613
+ "step": 868
2614
+ },
2615
+ {
2616
+ "epoch": 2.69,
2617
+ "learning_rate": 2e-05,
2618
+ "loss": 0.1753,
2619
+ "step": 870
2620
+ },
2621
+ {
2622
+ "epoch": 2.7,
2623
+ "learning_rate": 2e-05,
2624
+ "loss": 0.1739,
2625
+ "step": 872
2626
+ },
2627
+ {
2628
+ "epoch": 2.71,
2629
+ "learning_rate": 2e-05,
2630
+ "loss": 0.1752,
2631
+ "step": 874
2632
+ },
2633
+ {
2634
+ "epoch": 2.71,
2635
+ "learning_rate": 2e-05,
2636
+ "loss": 0.1748,
2637
+ "step": 876
2638
+ },
2639
+ {
2640
+ "epoch": 2.72,
2641
+ "learning_rate": 2e-05,
2642
+ "loss": 0.1813,
2643
+ "step": 878
2644
+ },
2645
+ {
2646
+ "epoch": 2.73,
2647
+ "learning_rate": 2e-05,
2648
+ "loss": 0.1745,
2649
+ "step": 880
2650
+ },
2651
+ {
2652
+ "epoch": 2.73,
2653
+ "learning_rate": 2e-05,
2654
+ "loss": 0.1737,
2655
+ "step": 882
2656
+ },
2657
+ {
2658
+ "epoch": 2.74,
2659
+ "learning_rate": 2e-05,
2660
+ "loss": 0.1769,
2661
+ "step": 884
2662
+ },
2663
+ {
2664
+ "epoch": 2.74,
2665
+ "learning_rate": 2e-05,
2666
+ "loss": 0.176,
2667
+ "step": 886
2668
+ },
2669
+ {
2670
+ "epoch": 2.75,
2671
+ "learning_rate": 2e-05,
2672
+ "loss": 0.1721,
2673
+ "step": 888
2674
+ },
2675
+ {
2676
+ "epoch": 2.76,
2677
+ "learning_rate": 2e-05,
2678
+ "loss": 0.174,
2679
+ "step": 890
2680
+ },
2681
+ {
2682
+ "epoch": 2.76,
2683
+ "learning_rate": 2e-05,
2684
+ "loss": 0.1713,
2685
+ "step": 892
2686
+ },
2687
+ {
2688
+ "epoch": 2.77,
2689
+ "learning_rate": 2e-05,
2690
+ "loss": 0.1702,
2691
+ "step": 894
2692
+ },
2693
+ {
2694
+ "epoch": 2.78,
2695
+ "learning_rate": 2e-05,
2696
+ "loss": 0.1749,
2697
+ "step": 896
2698
+ },
2699
+ {
2700
+ "epoch": 2.78,
2701
+ "learning_rate": 2e-05,
2702
+ "loss": 0.1737,
2703
+ "step": 898
2704
+ },
2705
+ {
2706
+ "epoch": 2.79,
2707
+ "learning_rate": 2e-05,
2708
+ "loss": 0.175,
2709
+ "step": 900
2710
+ },
2711
+ {
2712
+ "epoch": 2.79,
2713
+ "learning_rate": 2e-05,
2714
+ "loss": 0.1686,
2715
+ "step": 902
2716
+ },
2717
+ {
2718
+ "epoch": 2.8,
2719
+ "learning_rate": 2e-05,
2720
+ "loss": 0.1714,
2721
+ "step": 904
2722
+ },
2723
+ {
2724
+ "epoch": 2.81,
2725
+ "learning_rate": 2e-05,
2726
+ "loss": 0.1728,
2727
+ "step": 906
2728
+ },
2729
+ {
2730
+ "epoch": 2.81,
2731
+ "learning_rate": 2e-05,
2732
+ "loss": 0.1686,
2733
+ "step": 908
2734
+ },
2735
+ {
2736
+ "epoch": 2.82,
2737
+ "learning_rate": 2e-05,
2738
+ "loss": 0.1734,
2739
+ "step": 910
2740
+ },
2741
+ {
2742
+ "epoch": 2.82,
2743
+ "learning_rate": 2e-05,
2744
+ "loss": 0.1727,
2745
+ "step": 912
2746
+ },
2747
+ {
2748
+ "epoch": 2.83,
2749
+ "learning_rate": 2e-05,
2750
+ "loss": 0.1728,
2751
+ "step": 914
2752
+ },
2753
+ {
2754
+ "epoch": 2.84,
2755
+ "learning_rate": 2e-05,
2756
+ "loss": 0.1684,
2757
+ "step": 916
2758
+ },
2759
+ {
2760
+ "epoch": 2.84,
2761
+ "learning_rate": 2e-05,
2762
+ "loss": 0.17,
2763
+ "step": 918
2764
+ },
2765
+ {
2766
+ "epoch": 2.85,
2767
+ "learning_rate": 2e-05,
2768
+ "loss": 0.1664,
2769
+ "step": 920
2770
+ },
2771
+ {
2772
+ "epoch": 2.86,
2773
+ "learning_rate": 2e-05,
2774
+ "loss": 0.1657,
2775
+ "step": 922
2776
+ },
2777
+ {
2778
+ "epoch": 2.86,
2779
+ "learning_rate": 2e-05,
2780
+ "loss": 0.1695,
2781
+ "step": 924
2782
+ },
2783
+ {
2784
+ "epoch": 2.87,
2785
+ "learning_rate": 2e-05,
2786
+ "loss": 0.1731,
2787
+ "step": 926
2788
+ },
2789
+ {
2790
+ "epoch": 2.87,
2791
+ "learning_rate": 2e-05,
2792
+ "loss": 0.1681,
2793
+ "step": 928
2794
+ },
2795
+ {
2796
+ "epoch": 2.88,
2797
+ "learning_rate": 2e-05,
2798
+ "loss": 0.1683,
2799
+ "step": 930
2800
+ },
2801
+ {
2802
+ "epoch": 2.89,
2803
+ "learning_rate": 2e-05,
2804
+ "loss": 0.1685,
2805
+ "step": 932
2806
+ },
2807
+ {
2808
+ "epoch": 2.89,
2809
+ "learning_rate": 2e-05,
2810
+ "loss": 0.168,
2811
+ "step": 934
2812
+ },
2813
+ {
2814
+ "epoch": 2.9,
2815
+ "learning_rate": 2e-05,
2816
+ "loss": 0.1702,
2817
+ "step": 936
2818
+ },
2819
+ {
2820
+ "epoch": 2.91,
2821
+ "learning_rate": 2e-05,
2822
+ "loss": 0.1693,
2823
+ "step": 938
2824
+ },
2825
+ {
2826
+ "epoch": 2.91,
2827
+ "learning_rate": 2e-05,
2828
+ "loss": 0.1717,
2829
+ "step": 940
2830
+ },
2831
+ {
2832
+ "epoch": 2.92,
2833
+ "learning_rate": 2e-05,
2834
+ "loss": 0.1701,
2835
+ "step": 942
2836
+ },
2837
+ {
2838
+ "epoch": 2.92,
2839
+ "learning_rate": 2e-05,
2840
+ "loss": 0.1641,
2841
+ "step": 944
2842
+ },
2843
+ {
2844
+ "epoch": 2.93,
2845
+ "learning_rate": 2e-05,
2846
+ "loss": 0.1703,
2847
+ "step": 946
2848
+ },
2849
+ {
2850
+ "epoch": 2.94,
2851
+ "learning_rate": 2e-05,
2852
+ "loss": 0.1676,
2853
+ "step": 948
2854
+ },
2855
+ {
2856
+ "epoch": 2.94,
2857
+ "learning_rate": 2e-05,
2858
+ "loss": 0.1709,
2859
+ "step": 950
2860
+ },
2861
+ {
2862
+ "epoch": 2.95,
2863
+ "learning_rate": 2e-05,
2864
+ "loss": 0.1755,
2865
+ "step": 952
2866
+ },
2867
+ {
2868
+ "epoch": 2.95,
2869
+ "learning_rate": 2e-05,
2870
+ "loss": 0.1705,
2871
+ "step": 954
2872
+ },
2873
+ {
2874
+ "epoch": 2.96,
2875
+ "learning_rate": 2e-05,
2876
+ "loss": 0.1643,
2877
+ "step": 956
2878
+ },
2879
+ {
2880
+ "epoch": 2.97,
2881
+ "learning_rate": 2e-05,
2882
+ "loss": 0.1663,
2883
+ "step": 958
2884
+ },
2885
+ {
2886
+ "epoch": 2.97,
2887
+ "learning_rate": 2e-05,
2888
+ "loss": 0.1647,
2889
+ "step": 960
2890
+ },
2891
+ {
2892
+ "epoch": 2.98,
2893
+ "learning_rate": 2e-05,
2894
+ "loss": 0.1619,
2895
+ "step": 962
2896
+ },
2897
+ {
2898
+ "epoch": 2.99,
2899
+ "learning_rate": 2e-05,
2900
+ "loss": 0.17,
2901
+ "step": 964
2902
+ },
2903
+ {
2904
+ "epoch": 2.99,
2905
+ "learning_rate": 2e-05,
2906
+ "loss": 0.165,
2907
+ "step": 966
2908
+ },
2909
+ {
2910
+ "epoch": 3.0,
2911
+ "learning_rate": 2e-05,
2912
+ "loss": 0.1603,
2913
+ "step": 968
2914
+ },
2915
+ {
2916
+ "epoch": 3.0,
2917
+ "learning_rate": 2e-05,
2918
+ "loss": 0.1437,
2919
+ "step": 970
2920
+ },
2921
+ {
2922
+ "epoch": 3.01,
2923
+ "learning_rate": 2e-05,
2924
+ "loss": 0.1377,
2925
+ "step": 972
2926
+ },
2927
+ {
2928
+ "epoch": 3.02,
2929
+ "learning_rate": 2e-05,
2930
+ "loss": 0.1427,
2931
+ "step": 974
2932
+ },
2933
+ {
2934
+ "epoch": 3.02,
2935
+ "learning_rate": 2e-05,
2936
+ "loss": 0.148,
2937
+ "step": 976
2938
+ },
2939
+ {
2940
+ "epoch": 3.03,
2941
+ "learning_rate": 2e-05,
2942
+ "loss": 0.1383,
2943
+ "step": 978
2944
+ },
2945
+ {
2946
+ "epoch": 3.04,
2947
+ "learning_rate": 2e-05,
2948
+ "loss": 0.1441,
2949
+ "step": 980
2950
+ },
2951
+ {
2952
+ "epoch": 3.04,
2953
+ "learning_rate": 2e-05,
2954
+ "loss": 0.1416,
2955
+ "step": 982
2956
+ },
2957
+ {
2958
+ "epoch": 3.05,
2959
+ "learning_rate": 2e-05,
2960
+ "loss": 0.143,
2961
+ "step": 984
2962
+ },
2963
+ {
2964
+ "epoch": 3.05,
2965
+ "learning_rate": 2e-05,
2966
+ "loss": 0.1388,
2967
+ "step": 986
2968
+ },
2969
+ {
2970
+ "epoch": 3.06,
2971
+ "learning_rate": 2e-05,
2972
+ "loss": 0.1423,
2973
+ "step": 988
2974
+ },
2975
+ {
2976
+ "epoch": 3.07,
2977
+ "learning_rate": 2e-05,
2978
+ "loss": 0.1396,
2979
+ "step": 990
2980
+ },
2981
+ {
2982
+ "epoch": 3.07,
2983
+ "learning_rate": 2e-05,
2984
+ "loss": 0.1447,
2985
+ "step": 992
2986
+ },
2987
+ {
2988
+ "epoch": 3.08,
2989
+ "learning_rate": 2e-05,
2990
+ "loss": 0.1381,
2991
+ "step": 994
2992
+ },
2993
+ {
2994
+ "epoch": 3.08,
2995
+ "learning_rate": 2e-05,
2996
+ "loss": 0.1402,
2997
+ "step": 996
2998
+ },
2999
+ {
3000
+ "epoch": 3.09,
3001
+ "learning_rate": 2e-05,
3002
+ "loss": 0.1392,
3003
+ "step": 998
3004
+ },
3005
+ {
3006
+ "epoch": 3.1,
3007
+ "learning_rate": 2e-05,
3008
+ "loss": 0.1361,
3009
+ "step": 1000
3010
+ }
3011
+ ],
3012
+ "logging_steps": 2,
3013
+ "max_steps": 1288,
3014
+ "num_train_epochs": 4,
3015
+ "save_steps": 200,
3016
+ "total_flos": 2.2885242503168e+16,
3017
+ "trial_name": null,
3018
+ "trial_params": null
3019
+ }
training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d2aee8ac578286ab1de2c8f118e58f8dfbd556a31599d860520421328870178b
3
+ size 6840
zero_to_fp32.py ADDED
@@ -0,0 +1,578 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python
2
+
3
+ # Copyright (c) Microsoft Corporation.
4
+ # SPDX-License-Identifier: Apache-2.0
5
+
6
+ # DeepSpeed Team
7
+
8
+ # This script extracts fp32 consolidated weights from a zero 2 and 3 DeepSpeed checkpoints. It gets
9
+ # copied into the top level checkpoint dir, so the user can easily do the conversion at any point in
10
+ # the future. Once extracted, the weights don't require DeepSpeed and can be used in any
11
+ # application.
12
+ #
13
+ # example: python zero_to_fp32.py . pytorch_model.bin
14
+
15
+ import argparse
16
+ import torch
17
+ import glob
18
+ import math
19
+ import os
20
+ import re
21
+ from collections import OrderedDict
22
+ from dataclasses import dataclass
23
+
24
+ # while this script doesn't use deepspeed to recover data, since the checkpoints are pickled with
25
+ # DeepSpeed data structures it has to be available in the current python environment.
26
+ from deepspeed.utils import logger
27
+ from deepspeed.checkpoint.constants import (DS_VERSION, OPTIMIZER_STATE_DICT, SINGLE_PARTITION_OF_FP32_GROUPS,
28
+ FP32_FLAT_GROUPS, ZERO_STAGE, PARTITION_COUNT, PARAM_SHAPES, BUFFER_NAMES,
29
+ FROZEN_PARAM_SHAPES, FROZEN_PARAM_FRAGMENTS)
30
+
31
+
32
+ @dataclass
33
+ class zero_model_state:
34
+ buffers: dict()
35
+ param_shapes: dict()
36
+ shared_params: list
37
+ ds_version: int
38
+ frozen_param_shapes: dict()
39
+ frozen_param_fragments: dict()
40
+
41
+
42
+ debug = 0
43
+
44
+ # load to cpu
45
+ device = torch.device('cpu')
46
+
47
+
48
+ def atoi(text):
49
+ return int(text) if text.isdigit() else text
50
+
51
+
52
+ def natural_keys(text):
53
+ '''
54
+ alist.sort(key=natural_keys) sorts in human order
55
+ http://nedbatchelder.com/blog/200712/human_sorting.html
56
+ (See Toothy's implementation in the comments)
57
+ '''
58
+ return [atoi(c) for c in re.split(r'(\d+)', text)]
59
+
60
+
61
+ def get_model_state_file(checkpoint_dir, zero_stage):
62
+ if not os.path.isdir(checkpoint_dir):
63
+ raise FileNotFoundError(f"Directory '{checkpoint_dir}' doesn't exist")
64
+
65
+ # there should be only one file
66
+ if zero_stage == 2:
67
+ file = os.path.join(checkpoint_dir, "mp_rank_00_model_states.pt")
68
+ elif zero_stage == 3:
69
+ file = os.path.join(checkpoint_dir, "zero_pp_rank_0_mp_rank_00_model_states.pt")
70
+
71
+ if not os.path.exists(file):
72
+ raise FileNotFoundError(f"can't find model states file at '{file}'")
73
+
74
+ return file
75
+
76
+
77
+ def get_checkpoint_files(checkpoint_dir, glob_pattern):
78
+ # XXX: need to test that this simple glob rule works for multi-node setup too
79
+ ckpt_files = sorted(glob.glob(os.path.join(checkpoint_dir, glob_pattern)), key=natural_keys)
80
+
81
+ if len(ckpt_files) == 0:
82
+ raise FileNotFoundError(f"can't find {glob_pattern} files in directory '{checkpoint_dir}'")
83
+
84
+ return ckpt_files
85
+
86
+
87
+ def get_optim_files(checkpoint_dir):
88
+ return get_checkpoint_files(checkpoint_dir, "*_optim_states.pt")
89
+
90
+
91
+ def get_model_state_files(checkpoint_dir):
92
+ return get_checkpoint_files(checkpoint_dir, "*_model_states.pt")
93
+
94
+
95
+ def parse_model_states(files):
96
+ zero_model_states = []
97
+ for file in files:
98
+ state_dict = torch.load(file, map_location=device)
99
+
100
+ if BUFFER_NAMES not in state_dict:
101
+ raise ValueError(f"{file} is not a model state checkpoint")
102
+ buffer_names = state_dict[BUFFER_NAMES]
103
+ if debug:
104
+ print("Found buffers:", buffer_names)
105
+
106
+ # recover just the buffers while restoring them to fp32 if they were saved in fp16
107
+ buffers = {k: v.float() for k, v in state_dict["module"].items() if k in buffer_names}
108
+ param_shapes = state_dict[PARAM_SHAPES]
109
+
110
+ # collect parameters that are included in param_shapes
111
+ param_names = []
112
+ for s in param_shapes:
113
+ for name in s.keys():
114
+ param_names.append(name)
115
+
116
+ # update with frozen parameters
117
+ frozen_param_shapes = state_dict.get(FROZEN_PARAM_SHAPES, None)
118
+ if frozen_param_shapes is not None:
119
+ if debug:
120
+ print(f"Found frozen_param_shapes: {frozen_param_shapes}")
121
+ param_names += list(frozen_param_shapes.keys())
122
+
123
+ # handle shared params
124
+ shared_params = [[k, v] for k, v in state_dict["shared_params"].items()]
125
+
126
+ ds_version = state_dict.get(DS_VERSION, None)
127
+
128
+ frozen_param_fragments = state_dict.get(FROZEN_PARAM_FRAGMENTS, None)
129
+
130
+ z_model_state = zero_model_state(buffers=buffers,
131
+ param_shapes=param_shapes,
132
+ shared_params=shared_params,
133
+ ds_version=ds_version,
134
+ frozen_param_shapes=frozen_param_shapes,
135
+ frozen_param_fragments=frozen_param_fragments)
136
+ zero_model_states.append(z_model_state)
137
+
138
+ return zero_model_states
139
+
140
+
141
+ def parse_optim_states(files, ds_checkpoint_dir):
142
+
143
+ total_files = len(files)
144
+ state_dicts = []
145
+ for f in files:
146
+ state_dicts.append(torch.load(f, map_location=device))
147
+
148
+ if not ZERO_STAGE in state_dicts[0][OPTIMIZER_STATE_DICT]:
149
+ raise ValueError(f"{files[0]} is not a zero checkpoint")
150
+ zero_stage = state_dicts[0][OPTIMIZER_STATE_DICT][ZERO_STAGE]
151
+ world_size = state_dicts[0][OPTIMIZER_STATE_DICT][PARTITION_COUNT]
152
+
153
+ # For ZeRO-2 each param group can have different partition_count as data parallelism for expert
154
+ # parameters can be different from data parallelism for non-expert parameters. So we can just
155
+ # use the max of the partition_count to get the dp world_size.
156
+
157
+ if type(world_size) is list:
158
+ world_size = max(world_size)
159
+
160
+ if world_size != total_files:
161
+ raise ValueError(
162
+ f"Expected {world_size} of '*_optim_states.pt' under '{ds_checkpoint_dir}' but found {total_files} files. "
163
+ "Possibly due to an overwrite of an old checkpoint, or a checkpoint didn't get saved by one or more processes."
164
+ )
165
+
166
+ # the groups are named differently in each stage
167
+ if zero_stage == 2:
168
+ fp32_groups_key = SINGLE_PARTITION_OF_FP32_GROUPS
169
+ elif zero_stage == 3:
170
+ fp32_groups_key = FP32_FLAT_GROUPS
171
+ else:
172
+ raise ValueError(f"unknown zero stage {zero_stage}")
173
+
174
+ if zero_stage == 2:
175
+ fp32_flat_groups = [state_dicts[i][OPTIMIZER_STATE_DICT][fp32_groups_key] for i in range(len(state_dicts))]
176
+ elif zero_stage == 3:
177
+ # if there is more than one param group, there will be multiple flattened tensors - one
178
+ # flattened tensor per group - for simplicity merge them into a single tensor
179
+ #
180
+ # XXX: could make the script more memory efficient for when there are multiple groups - it
181
+ # will require matching the sub-lists of param_shapes for each param group flattened tensor
182
+
183
+ fp32_flat_groups = [
184
+ torch.cat(state_dicts[i][OPTIMIZER_STATE_DICT][fp32_groups_key], 0) for i in range(len(state_dicts))
185
+ ]
186
+
187
+ return zero_stage, world_size, fp32_flat_groups
188
+
189
+
190
+ def _get_fp32_state_dict_from_zero_checkpoint(ds_checkpoint_dir):
191
+ """
192
+ Returns fp32 state_dict reconstructed from ds checkpoint
193
+
194
+ Args:
195
+ - ``ds_checkpoint_dir``: path to the deepspeed checkpoint folder (where the optimizer files are)
196
+
197
+ """
198
+ print(f"Processing zero checkpoint '{ds_checkpoint_dir}'")
199
+
200
+ optim_files = get_optim_files(ds_checkpoint_dir)
201
+ zero_stage, world_size, fp32_flat_groups = parse_optim_states(optim_files, ds_checkpoint_dir)
202
+ print(f"Detected checkpoint of type zero stage {zero_stage}, world_size: {world_size}")
203
+
204
+ model_files = get_model_state_files(ds_checkpoint_dir)
205
+
206
+ zero_model_states = parse_model_states(model_files)
207
+ print(f'Parsing checkpoint created by deepspeed=={zero_model_states[0].ds_version}')
208
+
209
+ if zero_stage == 2:
210
+ return _get_fp32_state_dict_from_zero2_checkpoint(world_size, fp32_flat_groups, zero_model_states)
211
+ elif zero_stage == 3:
212
+ return _get_fp32_state_dict_from_zero3_checkpoint(world_size, fp32_flat_groups, zero_model_states)
213
+
214
+
215
+ def _zero2_merge_frozen_params(state_dict, zero_model_states):
216
+ if zero_model_states[0].frozen_param_shapes is None or len(zero_model_states[0].frozen_param_shapes) == 0:
217
+ return
218
+
219
+ frozen_param_shapes = zero_model_states[0].frozen_param_shapes
220
+ frozen_param_fragments = zero_model_states[0].frozen_param_fragments
221
+
222
+ if debug:
223
+ num_elem = sum(s.numel() for s in frozen_param_shapes.values())
224
+ print(f'rank 0: {FROZEN_PARAM_SHAPES}.numel = {num_elem}')
225
+
226
+ wanted_params = len(frozen_param_shapes)
227
+ wanted_numel = sum(s.numel() for s in frozen_param_shapes.values())
228
+ avail_numel = sum([p.numel() for p in frozen_param_fragments.values()])
229
+ print(f'Frozen params: Have {avail_numel} numels to process.')
230
+ print(f'Frozen params: Need {wanted_numel} numels in {wanted_params} params')
231
+
232
+ total_params = 0
233
+ total_numel = 0
234
+ for name, shape in frozen_param_shapes.items():
235
+ total_params += 1
236
+ unpartitioned_numel = shape.numel()
237
+ total_numel += unpartitioned_numel
238
+
239
+ state_dict[name] = frozen_param_fragments[name]
240
+
241
+ if debug:
242
+ print(f"{name} full shape: {shape} unpartitioned numel {unpartitioned_numel} ")
243
+
244
+ print(f"Reconstructed Frozen fp32 state dict with {total_params} params {total_numel} elements")
245
+
246
+
247
+ def _zero2_merge_trainable_params(state_dict, world_size, fp32_flat_groups, zero_model_states):
248
+ param_shapes = zero_model_states[0].param_shapes
249
+
250
+ # Reconstruction protocol:
251
+ #
252
+ # XXX: document this
253
+
254
+ if debug:
255
+ for i in range(world_size):
256
+ for j in range(len(fp32_flat_groups[0])):
257
+ print(f"{FP32_FLAT_GROUPS}[{i}][{j}].shape={fp32_flat_groups[i][j].shape}")
258
+
259
+ # XXX: memory usage doubles here (zero2)
260
+ num_param_groups = len(fp32_flat_groups[0])
261
+ merged_single_partition_of_fp32_groups = []
262
+ for i in range(num_param_groups):
263
+ merged_partitions = [sd[i] for sd in fp32_flat_groups]
264
+ full_single_fp32_vector = torch.cat(merged_partitions, 0)
265
+ merged_single_partition_of_fp32_groups.append(full_single_fp32_vector)
266
+ avail_numel = sum(
267
+ [full_single_fp32_vector.numel() for full_single_fp32_vector in merged_single_partition_of_fp32_groups])
268
+
269
+ if debug:
270
+ wanted_params = sum([len(shapes) for shapes in param_shapes])
271
+ wanted_numel = sum([sum(shape.numel() for shape in shapes.values()) for shapes in param_shapes])
272
+ # not asserting if there is a mismatch due to possible padding
273
+ print(f"Have {avail_numel} numels to process.")
274
+ print(f"Need {wanted_numel} numels in {wanted_params} params.")
275
+
276
+ # params
277
+ # XXX: for huge models that can't fit into the host's RAM we will have to recode this to support
278
+ # out-of-core computing solution
279
+ total_numel = 0
280
+ total_params = 0
281
+ for shapes, full_single_fp32_vector in zip(param_shapes, merged_single_partition_of_fp32_groups):
282
+ offset = 0
283
+ avail_numel = full_single_fp32_vector.numel()
284
+ for name, shape in shapes.items():
285
+
286
+ unpartitioned_numel = shape.numel()
287
+ total_numel += unpartitioned_numel
288
+ total_params += 1
289
+
290
+ if debug:
291
+ print(f"{name} full shape: {shape} unpartitioned numel {unpartitioned_numel} ")
292
+ state_dict[name] = full_single_fp32_vector.narrow(0, offset, unpartitioned_numel).view(shape)
293
+ offset += unpartitioned_numel
294
+
295
+ # Z2 started to align to 2*world_size to improve nccl performance. Therefore both offset and
296
+ # avail_numel can differ by anywhere between 0..2*world_size. Due to two unrelated complex
297
+ # paddings performed in the code it's almost impossible to predict the exact numbers w/o the
298
+ # live optimizer object, so we are checking that the numbers are within the right range
299
+ align_to = 2 * world_size
300
+
301
+ def zero2_align(x):
302
+ return align_to * math.ceil(x / align_to)
303
+
304
+ if debug:
305
+ print(f"original offset={offset}, avail_numel={avail_numel}")
306
+
307
+ offset = zero2_align(offset)
308
+ avail_numel = zero2_align(avail_numel)
309
+
310
+ if debug:
311
+ print(f"aligned offset={offset}, avail_numel={avail_numel}")
312
+
313
+ # Sanity check
314
+ if offset != avail_numel:
315
+ raise ValueError(f"consumed {offset} numels out of {avail_numel} - something is wrong")
316
+
317
+ print(f"Reconstructed fp32 state dict with {total_params} params {total_numel} elements")
318
+
319
+
320
+ def _get_fp32_state_dict_from_zero2_checkpoint(world_size, fp32_flat_groups, zero_model_states):
321
+ state_dict = OrderedDict()
322
+
323
+ # buffers
324
+ buffers = zero_model_states[0].buffers
325
+ state_dict.update(buffers)
326
+ if debug:
327
+ print(f"added {len(buffers)} buffers")
328
+
329
+ _zero2_merge_frozen_params(state_dict, zero_model_states)
330
+
331
+ _zero2_merge_trainable_params(state_dict, world_size, fp32_flat_groups, zero_model_states)
332
+
333
+ # recover shared parameters
334
+ for pair in zero_model_states[0].shared_params:
335
+ if pair[1] in state_dict:
336
+ state_dict[pair[0]] = state_dict[pair[1]]
337
+
338
+ return state_dict
339
+
340
+
341
+ def zero3_partitioned_param_info(unpartitioned_numel, world_size):
342
+ remainder = unpartitioned_numel % world_size
343
+ padding_numel = (world_size - remainder) if remainder else 0
344
+ partitioned_numel = math.ceil(unpartitioned_numel / world_size)
345
+ return partitioned_numel, padding_numel
346
+
347
+
348
+ def _zero3_merge_frozen_params(state_dict, world_size, zero_model_states):
349
+ if zero_model_states[0].frozen_param_shapes is None or len(zero_model_states[0].frozen_param_shapes) == 0:
350
+ return
351
+
352
+ if debug:
353
+ for i in range(world_size):
354
+ num_elem = sum(s.numel() for s in zero_model_states[i].frozen_param_fragments.values())
355
+ print(f'rank {i}: {FROZEN_PARAM_SHAPES}.numel = {num_elem}')
356
+
357
+ frozen_param_shapes = zero_model_states[0].frozen_param_shapes
358
+ wanted_params = len(frozen_param_shapes)
359
+ wanted_numel = sum(s.numel() for s in frozen_param_shapes.values())
360
+ avail_numel = sum([p.numel() for p in zero_model_states[0].frozen_param_fragments.values()]) * world_size
361
+ print(f'Frozen params: Have {avail_numel} numels to process.')
362
+ print(f'Frozen params: Need {wanted_numel} numels in {wanted_params} params')
363
+
364
+ total_params = 0
365
+ total_numel = 0
366
+ for name, shape in zero_model_states[0].frozen_param_shapes.items():
367
+ total_params += 1
368
+ unpartitioned_numel = shape.numel()
369
+ total_numel += unpartitioned_numel
370
+
371
+ param_frags = tuple(model_state.frozen_param_fragments[name] for model_state in zero_model_states)
372
+ state_dict[name] = torch.cat(param_frags, 0).narrow(0, 0, unpartitioned_numel).view(shape)
373
+
374
+ partitioned_numel, partitioned_padding_numel = zero3_partitioned_param_info(unpartitioned_numel, world_size)
375
+
376
+ if debug:
377
+ print(
378
+ f"Frozen params: {total_params} {name} full shape: {shape} partition0 numel={partitioned_numel} partitioned_padding_numel={partitioned_padding_numel}"
379
+ )
380
+
381
+ print(f"Reconstructed Frozen fp32 state dict with {total_params} params {total_numel} elements")
382
+
383
+
384
+ def _zero3_merge_trainable_params(state_dict, world_size, fp32_flat_groups, zero_model_states):
385
+ param_shapes = zero_model_states[0].param_shapes
386
+ avail_numel = fp32_flat_groups[0].numel() * world_size
387
+ # Reconstruction protocol: For zero3 we need to zip the partitions together at boundary of each
388
+ # param, re-consolidating each param, while dealing with padding if any
389
+
390
+ # merge list of dicts, preserving order
391
+ param_shapes = {k: v for d in param_shapes for k, v in d.items()}
392
+
393
+ if debug:
394
+ for i in range(world_size):
395
+ print(f"{FP32_FLAT_GROUPS}[{i}].shape={fp32_flat_groups[i].shape}")
396
+
397
+ wanted_params = len(param_shapes)
398
+ wanted_numel = sum(shape.numel() for shape in param_shapes.values())
399
+ # not asserting if there is a mismatch due to possible padding
400
+ avail_numel = fp32_flat_groups[0].numel() * world_size
401
+ print(f"Trainable params: Have {avail_numel} numels to process.")
402
+ print(f"Trainable params: Need {wanted_numel} numels in {wanted_params} params.")
403
+
404
+ # params
405
+ # XXX: for huge models that can't fit into the host's RAM we will have to recode this to support
406
+ # out-of-core computing solution
407
+ offset = 0
408
+ total_numel = 0
409
+ total_params = 0
410
+ for name, shape in param_shapes.items():
411
+
412
+ unpartitioned_numel = shape.numel()
413
+ total_numel += unpartitioned_numel
414
+ total_params += 1
415
+
416
+ partitioned_numel, partitioned_padding_numel = zero3_partitioned_param_info(unpartitioned_numel, world_size)
417
+
418
+ if debug:
419
+ print(
420
+ f"Trainable params: {total_params} {name} full shape: {shape} partition0 numel={partitioned_numel} partitioned_padding_numel={partitioned_padding_numel}"
421
+ )
422
+
423
+ # XXX: memory usage doubles here
424
+ state_dict[name] = torch.cat(
425
+ tuple(fp32_flat_groups[i].narrow(0, offset, partitioned_numel) for i in range(world_size)),
426
+ 0).narrow(0, 0, unpartitioned_numel).view(shape)
427
+ offset += partitioned_numel
428
+
429
+ offset *= world_size
430
+
431
+ # Sanity check
432
+ if offset != avail_numel:
433
+ raise ValueError(f"consumed {offset} numels out of {avail_numel} - something is wrong")
434
+
435
+ print(f"Reconstructed Trainable fp32 state dict with {total_params} params {total_numel} elements")
436
+
437
+
438
+ def _get_fp32_state_dict_from_zero3_checkpoint(world_size, fp32_flat_groups, zero_model_states):
439
+ state_dict = OrderedDict()
440
+
441
+ # buffers
442
+ buffers = zero_model_states[0].buffers
443
+ state_dict.update(buffers)
444
+ if debug:
445
+ print(f"added {len(buffers)} buffers")
446
+
447
+ _zero3_merge_frozen_params(state_dict, world_size, zero_model_states)
448
+
449
+ _zero3_merge_trainable_params(state_dict, world_size, fp32_flat_groups, zero_model_states)
450
+
451
+ # recover shared parameters
452
+ for pair in zero_model_states[0].shared_params:
453
+ if pair[1] in state_dict:
454
+ state_dict[pair[0]] = state_dict[pair[1]]
455
+
456
+ return state_dict
457
+
458
+
459
+ def get_fp32_state_dict_from_zero_checkpoint(checkpoint_dir, tag=None):
460
+ """
461
+ Convert ZeRO 2 or 3 checkpoint into a single fp32 consolidated state_dict that can be loaded with
462
+ ``load_state_dict()`` and used for training without DeepSpeed or shared with others, for example
463
+ via a model hub.
464
+
465
+ Args:
466
+ - ``checkpoint_dir``: path to the desired checkpoint folder
467
+ - ``tag``: checkpoint tag used as a unique identifier for checkpoint. If not provided will attempt to load tag in 'latest' file. e.g., ``global_step14``
468
+
469
+ Returns:
470
+ - pytorch ``state_dict``
471
+
472
+ Note: this approach may not work if your application doesn't have sufficient free CPU memory and
473
+ you may need to use the offline approach using the ``zero_to_fp32.py`` script that is saved with
474
+ the checkpoint.
475
+
476
+ A typical usage might be ::
477
+
478
+ from deepspeed.utils.zero_to_fp32 import get_fp32_state_dict_from_zero_checkpoint
479
+ # do the training and checkpoint saving
480
+ state_dict = get_fp32_state_dict_from_zero_checkpoint(checkpoint_dir) # already on cpu
481
+ model = model.cpu() # move to cpu
482
+ model.load_state_dict(state_dict)
483
+ # submit to model hub or save the model to share with others
484
+
485
+ In this example the ``model`` will no longer be usable in the deepspeed context of the same
486
+ application. i.e. you will need to re-initialize the deepspeed engine, since
487
+ ``model.load_state_dict(state_dict)`` will remove all the deepspeed magic from it.
488
+
489
+ If you want it all done for you, use ``load_state_dict_from_zero_checkpoint`` instead.
490
+
491
+ """
492
+ if tag is None:
493
+ latest_path = os.path.join(checkpoint_dir, 'latest')
494
+ if os.path.isfile(latest_path):
495
+ with open(latest_path, 'r') as fd:
496
+ tag = fd.read().strip()
497
+ else:
498
+ raise ValueError(f"Unable to find 'latest' file at {latest_path}")
499
+
500
+ ds_checkpoint_dir = os.path.join(checkpoint_dir, tag)
501
+
502
+ if not os.path.isdir(ds_checkpoint_dir):
503
+ raise FileNotFoundError(f"Directory '{ds_checkpoint_dir}' doesn't exist")
504
+
505
+ return _get_fp32_state_dict_from_zero_checkpoint(ds_checkpoint_dir)
506
+
507
+
508
+ def convert_zero_checkpoint_to_fp32_state_dict(checkpoint_dir, output_file, tag=None):
509
+ """
510
+ Convert ZeRO 2 or 3 checkpoint into a single fp32 consolidated ``state_dict`` file that can be
511
+ loaded with ``torch.load(file)`` + ``load_state_dict()`` and used for training without DeepSpeed.
512
+
513
+ Args:
514
+ - ``checkpoint_dir``: path to the desired checkpoint folder. (one that contains the tag-folder, like ``global_step14``)
515
+ - ``output_file``: path to the pytorch fp32 state_dict output file (e.g. path/pytorch_model.bin)
516
+ - ``tag``: checkpoint tag used as a unique identifier for checkpoint. If not provided will attempt to load tag in the file named ``latest`` in the checkpoint folder, e.g., ``global_step14``
517
+ """
518
+
519
+ state_dict = get_fp32_state_dict_from_zero_checkpoint(checkpoint_dir, tag)
520
+ print(f"Saving fp32 state dict to {output_file}")
521
+ torch.save(state_dict, output_file)
522
+
523
+
524
+ def load_state_dict_from_zero_checkpoint(model, checkpoint_dir, tag=None):
525
+ """
526
+ 1. Put the provided model to cpu
527
+ 2. Convert ZeRO 2 or 3 checkpoint into a single fp32 consolidated ``state_dict``
528
+ 3. Load it into the provided model
529
+
530
+ Args:
531
+ - ``model``: the model object to update
532
+ - ``checkpoint_dir``: path to the desired checkpoint folder. (one that contains the tag-folder, like ``global_step14``)
533
+ - ``tag``: checkpoint tag used as a unique identifier for checkpoint. If not provided will attempt to load tag in the file named ``latest`` in the checkpoint folder, e.g., ``global_step14``
534
+
535
+ Returns:
536
+ - ``model`: modified model
537
+
538
+ Make sure you have plenty of CPU memory available before you call this function. If you don't
539
+ have enough use the ``zero_to_fp32.py`` utility to do the conversion. You will find it
540
+ conveniently placed for you in the checkpoint folder.
541
+
542
+ A typical usage might be ::
543
+
544
+ from deepspeed.utils.zero_to_fp32 import load_state_dict_from_zero_checkpoint
545
+ model = load_state_dict_from_zero_checkpoint(trainer.model, checkpoint_dir)
546
+ # submit to model hub or save the model to share with others
547
+
548
+ Note, that once this was run, the ``model`` will no longer be usable in the deepspeed context
549
+ of the same application. i.e. you will need to re-initialize the deepspeed engine, since
550
+ ``model.load_state_dict(state_dict)`` will remove all the deepspeed magic from it.
551
+
552
+ """
553
+ logger.info(f"Extracting fp32 weights")
554
+ state_dict = get_fp32_state_dict_from_zero_checkpoint(checkpoint_dir, tag)
555
+
556
+ logger.info(f"Overwriting model with fp32 weights")
557
+ model = model.cpu()
558
+ model.load_state_dict(state_dict, strict=False)
559
+
560
+ return model
561
+
562
+
563
+ if __name__ == "__main__":
564
+
565
+ parser = argparse.ArgumentParser()
566
+ parser.add_argument("checkpoint_dir",
567
+ type=str,
568
+ help="path to the desired checkpoint folder, e.g., path/checkpoint-12")
569
+ parser.add_argument(
570
+ "output_file",
571
+ type=str,
572
+ help="path to the pytorch fp32 state_dict output file (e.g. path/checkpoint-12/pytorch_model.bin)")
573
+ parser.add_argument("-d", "--debug", action='store_true', help="enable debug")
574
+ args = parser.parse_args()
575
+
576
+ debug = args.debug
577
+
578
+ convert_zero_checkpoint_to_fp32_state_dict(args.checkpoint_dir, args.output_file)