vsarathy commited on
Commit
340eb92
1 Parent(s): 1576055

Training in progress, epoch 0

Browse files
adapter_config.json CHANGED
@@ -12,14 +12,14 @@
12
  "lora_dropout": 0.05,
13
  "modules_to_save": null,
14
  "peft_type": "LORA",
15
- "r": 64,
16
  "rank_pattern": {},
17
  "revision": null,
18
  "target_modules": [
19
- "dense",
20
- "query_key_value",
21
  "dense_h_to_4h",
22
- "dense_4h_to_h"
 
 
23
  ],
24
  "task_type": "CAUSAL_LM"
25
  }
 
12
  "lora_dropout": 0.05,
13
  "modules_to_save": null,
14
  "peft_type": "LORA",
15
+ "r": 32,
16
  "rank_pattern": {},
17
  "revision": null,
18
  "target_modules": [
 
 
19
  "dense_h_to_4h",
20
+ "query_key_value",
21
+ "dense_4h_to_h",
22
+ "dense"
23
  ],
24
  "task_type": "CAUSAL_LM"
25
  }
adapter_model.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d3b794f975f5e8d0346ee9b42adf3bd4dfd2aa206dd730bff8be88c0041ddf14
3
- size 522284877
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9b590af98d38b0cba5ffca897f94ff33198d51e1b6e0a1b94aba0f00222c1890
3
+ size 261189453
config.json CHANGED
@@ -21,6 +21,7 @@
21
  "hidden_size": 4544,
22
  "initializer_range": 0.02,
23
  "layer_norm_epsilon": 1e-05,
 
24
  "model_type": "falcon",
25
  "multi_query": true,
26
  "new_decoder_architecture": false,
@@ -40,8 +41,10 @@
40
  "load_in_8bit": false,
41
  "quant_method": "bitsandbytes"
42
  },
 
 
43
  "torch_dtype": "bfloat16",
44
- "transformers_version": "4.34.0.dev0",
45
  "use_cache": false,
46
- "vocab_size": 65024
47
  }
 
21
  "hidden_size": 4544,
22
  "initializer_range": 0.02,
23
  "layer_norm_epsilon": 1e-05,
24
+ "max_position_embeddings": 2048,
25
  "model_type": "falcon",
26
  "multi_query": true,
27
  "new_decoder_architecture": false,
 
41
  "load_in_8bit": false,
42
  "quant_method": "bitsandbytes"
43
  },
44
+ "rope_scaling": null,
45
+ "rope_theta": 10000.0,
46
  "torch_dtype": "bfloat16",
47
+ "transformers_version": "4.35.0.dev0",
48
  "use_cache": false,
49
+ "vocab_size": 65027
50
  }
special_tokens_map.json CHANGED
@@ -12,7 +12,8 @@
12
  ">>SUFFIX<<",
13
  ">>MIDDLE<<"
14
  ],
15
- "bos_token": ">>ABSTRACT<<",
16
- "eos_token": "<|endoftext|>",
17
- "pad_token": "<|endoftext|>"
 
18
  }
 
12
  ">>SUFFIX<<",
13
  ">>MIDDLE<<"
14
  ],
15
+ "bos_token": "<s>",
16
+ "eos_token": "</s>",
17
+ "pad_token": "</s>",
18
+ "unk_token": "<unk>"
19
  }
tokenizer.json CHANGED
@@ -7,8 +7,8 @@
7
  "id": 0,
8
  "content": ">>TITLE<<",
9
  "single_word": false,
10
- "lstrip": true,
11
- "rstrip": true,
12
  "normalized": false,
13
  "special": true
14
  },
@@ -16,8 +16,8 @@
16
  "id": 1,
17
  "content": ">>ABSTRACT<<",
18
  "single_word": false,
19
- "lstrip": true,
20
- "rstrip": true,
21
  "normalized": false,
22
  "special": true
23
  },
@@ -25,8 +25,8 @@
25
  "id": 2,
26
  "content": ">>INTRODUCTION<<",
27
  "single_word": false,
28
- "lstrip": true,
29
- "rstrip": true,
30
  "normalized": false,
31
  "special": true
32
  },
@@ -34,8 +34,8 @@
34
  "id": 3,
35
  "content": ">>SUMMARY<<",
36
  "single_word": false,
37
- "lstrip": true,
38
- "rstrip": true,
39
  "normalized": false,
40
  "special": true
41
  },
@@ -43,8 +43,8 @@
43
  "id": 4,
44
  "content": ">>COMMENT<<",
45
  "single_word": false,
46
- "lstrip": true,
47
- "rstrip": true,
48
  "normalized": false,
49
  "special": true
50
  },
@@ -52,8 +52,8 @@
52
  "id": 5,
53
  "content": ">>ANSWER<<",
54
  "single_word": false,
55
- "lstrip": true,
56
- "rstrip": true,
57
  "normalized": false,
58
  "special": true
59
  },
@@ -61,8 +61,8 @@
61
  "id": 6,
62
  "content": ">>QUESTION<<",
63
  "single_word": false,
64
- "lstrip": true,
65
- "rstrip": true,
66
  "normalized": false,
67
  "special": true
68
  },
@@ -70,8 +70,8 @@
70
  "id": 7,
71
  "content": ">>DOMAIN<<",
72
  "single_word": false,
73
- "lstrip": true,
74
- "rstrip": true,
75
  "normalized": false,
76
  "special": true
77
  },
@@ -79,8 +79,8 @@
79
  "id": 8,
80
  "content": ">>PREFIX<<",
81
  "single_word": false,
82
- "lstrip": true,
83
- "rstrip": true,
84
  "normalized": false,
85
  "special": true
86
  },
@@ -88,8 +88,8 @@
88
  "id": 9,
89
  "content": ">>SUFFIX<<",
90
  "single_word": false,
91
- "lstrip": true,
92
- "rstrip": true,
93
  "normalized": false,
94
  "special": true
95
  },
@@ -97,8 +97,8 @@
97
  "id": 10,
98
  "content": ">>MIDDLE<<",
99
  "single_word": false,
100
- "lstrip": true,
101
- "rstrip": true,
102
  "normalized": false,
103
  "special": true
104
  },
@@ -106,6 +106,33 @@
106
  "id": 11,
107
  "content": "<|endoftext|>",
108
  "single_word": false,
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
109
  "lstrip": true,
110
  "rstrip": true,
111
  "normalized": false,
 
7
  "id": 0,
8
  "content": ">>TITLE<<",
9
  "single_word": false,
10
+ "lstrip": false,
11
+ "rstrip": false,
12
  "normalized": false,
13
  "special": true
14
  },
 
16
  "id": 1,
17
  "content": ">>ABSTRACT<<",
18
  "single_word": false,
19
+ "lstrip": false,
20
+ "rstrip": false,
21
  "normalized": false,
22
  "special": true
23
  },
 
25
  "id": 2,
26
  "content": ">>INTRODUCTION<<",
27
  "single_word": false,
28
+ "lstrip": false,
29
+ "rstrip": false,
30
  "normalized": false,
31
  "special": true
32
  },
 
34
  "id": 3,
35
  "content": ">>SUMMARY<<",
36
  "single_word": false,
37
+ "lstrip": false,
38
+ "rstrip": false,
39
  "normalized": false,
40
  "special": true
41
  },
 
43
  "id": 4,
44
  "content": ">>COMMENT<<",
45
  "single_word": false,
46
+ "lstrip": false,
47
+ "rstrip": false,
48
  "normalized": false,
49
  "special": true
50
  },
 
52
  "id": 5,
53
  "content": ">>ANSWER<<",
54
  "single_word": false,
55
+ "lstrip": false,
56
+ "rstrip": false,
57
  "normalized": false,
58
  "special": true
59
  },
 
61
  "id": 6,
62
  "content": ">>QUESTION<<",
63
  "single_word": false,
64
+ "lstrip": false,
65
+ "rstrip": false,
66
  "normalized": false,
67
  "special": true
68
  },
 
70
  "id": 7,
71
  "content": ">>DOMAIN<<",
72
  "single_word": false,
73
+ "lstrip": false,
74
+ "rstrip": false,
75
  "normalized": false,
76
  "special": true
77
  },
 
79
  "id": 8,
80
  "content": ">>PREFIX<<",
81
  "single_word": false,
82
+ "lstrip": false,
83
+ "rstrip": false,
84
  "normalized": false,
85
  "special": true
86
  },
 
88
  "id": 9,
89
  "content": ">>SUFFIX<<",
90
  "single_word": false,
91
+ "lstrip": false,
92
+ "rstrip": false,
93
  "normalized": false,
94
  "special": true
95
  },
 
97
  "id": 10,
98
  "content": ">>MIDDLE<<",
99
  "single_word": false,
100
+ "lstrip": false,
101
+ "rstrip": false,
102
  "normalized": false,
103
  "special": true
104
  },
 
106
  "id": 11,
107
  "content": "<|endoftext|>",
108
  "single_word": false,
109
+ "lstrip": false,
110
+ "rstrip": false,
111
+ "normalized": false,
112
+ "special": true
113
+ },
114
+ {
115
+ "id": 65024,
116
+ "content": "<s>",
117
+ "single_word": false,
118
+ "lstrip": true,
119
+ "rstrip": true,
120
+ "normalized": false,
121
+ "special": true
122
+ },
123
+ {
124
+ "id": 65025,
125
+ "content": "</s>",
126
+ "single_word": false,
127
+ "lstrip": true,
128
+ "rstrip": true,
129
+ "normalized": false,
130
+ "special": true
131
+ },
132
+ {
133
+ "id": 65026,
134
+ "content": "<unk>",
135
+ "single_word": false,
136
  "lstrip": true,
137
  "rstrip": true,
138
  "normalized": false,
tokenizer_config.json CHANGED
@@ -3,94 +3,118 @@
3
  "added_tokens_decoder": {
4
  "0": {
5
  "content": ">>TITLE<<",
6
- "lstrip": true,
7
  "normalized": false,
8
- "rstrip": true,
9
  "single_word": false,
10
  "special": true
11
  },
12
  "1": {
13
  "content": ">>ABSTRACT<<",
14
- "lstrip": true,
15
  "normalized": false,
16
- "rstrip": true,
17
  "single_word": false,
18
  "special": true
19
  },
20
  "2": {
21
  "content": ">>INTRODUCTION<<",
22
- "lstrip": true,
23
  "normalized": false,
24
- "rstrip": true,
25
  "single_word": false,
26
  "special": true
27
  },
28
  "3": {
29
  "content": ">>SUMMARY<<",
30
- "lstrip": true,
31
  "normalized": false,
32
- "rstrip": true,
33
  "single_word": false,
34
  "special": true
35
  },
36
  "4": {
37
  "content": ">>COMMENT<<",
38
- "lstrip": true,
39
  "normalized": false,
40
- "rstrip": true,
41
  "single_word": false,
42
  "special": true
43
  },
44
  "5": {
45
  "content": ">>ANSWER<<",
46
- "lstrip": true,
47
  "normalized": false,
48
- "rstrip": true,
49
  "single_word": false,
50
  "special": true
51
  },
52
  "6": {
53
  "content": ">>QUESTION<<",
54
- "lstrip": true,
55
  "normalized": false,
56
- "rstrip": true,
57
  "single_word": false,
58
  "special": true
59
  },
60
  "7": {
61
  "content": ">>DOMAIN<<",
62
- "lstrip": true,
63
  "normalized": false,
64
- "rstrip": true,
65
  "single_word": false,
66
  "special": true
67
  },
68
  "8": {
69
  "content": ">>PREFIX<<",
70
- "lstrip": true,
71
  "normalized": false,
72
- "rstrip": true,
73
  "single_word": false,
74
  "special": true
75
  },
76
  "9": {
77
  "content": ">>SUFFIX<<",
78
- "lstrip": true,
79
  "normalized": false,
80
- "rstrip": true,
81
  "single_word": false,
82
  "special": true
83
  },
84
  "10": {
85
  "content": ">>MIDDLE<<",
86
- "lstrip": true,
87
  "normalized": false,
88
- "rstrip": true,
89
  "single_word": false,
90
  "special": true
91
  },
92
  "11": {
93
  "content": "<|endoftext|>",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
94
  "lstrip": true,
95
  "normalized": false,
96
  "rstrip": true,
@@ -112,7 +136,7 @@
112
  ">>MIDDLE<<"
113
  ],
114
  "clean_up_tokenization_spaces": true,
115
- "eos_token": "<|endoftext|>",
116
  "model_input_names": [
117
  "input_ids",
118
  "attention_mask"
 
3
  "added_tokens_decoder": {
4
  "0": {
5
  "content": ">>TITLE<<",
6
+ "lstrip": false,
7
  "normalized": false,
8
+ "rstrip": false,
9
  "single_word": false,
10
  "special": true
11
  },
12
  "1": {
13
  "content": ">>ABSTRACT<<",
14
+ "lstrip": false,
15
  "normalized": false,
16
+ "rstrip": false,
17
  "single_word": false,
18
  "special": true
19
  },
20
  "2": {
21
  "content": ">>INTRODUCTION<<",
22
+ "lstrip": false,
23
  "normalized": false,
24
+ "rstrip": false,
25
  "single_word": false,
26
  "special": true
27
  },
28
  "3": {
29
  "content": ">>SUMMARY<<",
30
+ "lstrip": false,
31
  "normalized": false,
32
+ "rstrip": false,
33
  "single_word": false,
34
  "special": true
35
  },
36
  "4": {
37
  "content": ">>COMMENT<<",
38
+ "lstrip": false,
39
  "normalized": false,
40
+ "rstrip": false,
41
  "single_word": false,
42
  "special": true
43
  },
44
  "5": {
45
  "content": ">>ANSWER<<",
46
+ "lstrip": false,
47
  "normalized": false,
48
+ "rstrip": false,
49
  "single_word": false,
50
  "special": true
51
  },
52
  "6": {
53
  "content": ">>QUESTION<<",
54
+ "lstrip": false,
55
  "normalized": false,
56
+ "rstrip": false,
57
  "single_word": false,
58
  "special": true
59
  },
60
  "7": {
61
  "content": ">>DOMAIN<<",
62
+ "lstrip": false,
63
  "normalized": false,
64
+ "rstrip": false,
65
  "single_word": false,
66
  "special": true
67
  },
68
  "8": {
69
  "content": ">>PREFIX<<",
70
+ "lstrip": false,
71
  "normalized": false,
72
+ "rstrip": false,
73
  "single_word": false,
74
  "special": true
75
  },
76
  "9": {
77
  "content": ">>SUFFIX<<",
78
+ "lstrip": false,
79
  "normalized": false,
80
+ "rstrip": false,
81
  "single_word": false,
82
  "special": true
83
  },
84
  "10": {
85
  "content": ">>MIDDLE<<",
86
+ "lstrip": false,
87
  "normalized": false,
88
+ "rstrip": false,
89
  "single_word": false,
90
  "special": true
91
  },
92
  "11": {
93
  "content": "<|endoftext|>",
94
+ "lstrip": false,
95
+ "normalized": false,
96
+ "rstrip": false,
97
+ "single_word": false,
98
+ "special": true
99
+ },
100
+ "65024": {
101
+ "content": "<s>",
102
+ "lstrip": true,
103
+ "normalized": false,
104
+ "rstrip": true,
105
+ "single_word": false,
106
+ "special": true
107
+ },
108
+ "65025": {
109
+ "content": "</s>",
110
+ "lstrip": true,
111
+ "normalized": false,
112
+ "rstrip": true,
113
+ "single_word": false,
114
+ "special": true
115
+ },
116
+ "65026": {
117
+ "content": "<unk>",
118
  "lstrip": true,
119
  "normalized": false,
120
  "rstrip": true,
 
136
  ">>MIDDLE<<"
137
  ],
138
  "clean_up_tokenization_spaces": true,
139
+ "eos_token": "</s>",
140
  "model_input_names": [
141
  "input_ids",
142
  "attention_mask"
training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:80d4c6c13aa424e69e699ca029f25025f3867fadd4c6058fa91d42435b9a0b84
3
- size 4475
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:72f4ab0bc1682946ae88003cc1cf4c559eca058b2c8743584d85f016de994137
3
+ size 4539