gapendl commited on
Commit
3476051
·
verified ·
1 Parent(s): ff358d0

finetuned qa dataset 4 epochs

Browse files
Files changed (4) hide show
  1. README.md +3 -3
  2. config.json +1 -1
  3. model.safetensors +1 -1
  4. tokenizer_config.json +7 -0
README.md CHANGED
@@ -86,9 +86,9 @@ The model was trained with the parameters:
86
 
87
  **DataLoader**:
88
 
89
- `torch.utils.data.dataloader.DataLoader` of length 773 with parameters:
90
  ```
91
- {'batch_size': 128, 'sampler': 'torch.utils.data.sampler.RandomSampler', 'batch_sampler': 'torch.utils.data.sampler.BatchSampler'}
92
  ```
93
 
94
  **Loss**:
@@ -111,7 +111,7 @@ Parameters of the fit()-Method:
111
  },
112
  "scheduler": "WarmupLinear",
113
  "steps_per_epoch": null,
114
- "warmup_steps": 309,
115
  "weight_decay": 0.01
116
  }
117
  ```
 
86
 
87
  **DataLoader**:
88
 
89
+ `torch.utils.data.dataloader.DataLoader` of length 2062 with parameters:
90
  ```
91
+ {'batch_size': 48, 'sampler': 'torch.utils.data.sampler.RandomSampler', 'batch_sampler': 'torch.utils.data.sampler.BatchSampler'}
92
  ```
93
 
94
  **Loss**:
 
111
  },
112
  "scheduler": "WarmupLinear",
113
  "steps_per_epoch": null,
114
+ "warmup_steps": 824,
115
  "weight_decay": 0.01
116
  }
117
  ```
config.json CHANGED
@@ -1,5 +1,5 @@
1
  {
2
- "_name_or_path": "gapendl/gbert-base-kurier",
3
  "architectures": [
4
  "BertModel"
5
  ],
 
1
  {
2
+ "_name_or_path": "gapendl/geronimo-base",
3
  "architectures": [
4
  "BertModel"
5
  ],
model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:19f6b1b578caf868cdbd09982fff8a844cd7ec5fdc17631b0f53b823175a5820
3
  size 439733088
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:27c9e463bd0fabd102a9f1689afecfb4d38db14b44ecb44526643ad30e0ecb22
3
  size 439733088
tokenizer_config.json CHANGED
@@ -47,12 +47,19 @@
47
  "do_lower_case": false,
48
  "mask_token": "[MASK]",
49
  "max_len": 512,
 
50
  "model_max_length": 512,
51
  "never_split": null,
 
52
  "pad_token": "[PAD]",
 
 
53
  "sep_token": "[SEP]",
 
54
  "strip_accents": false,
55
  "tokenize_chinese_chars": true,
56
  "tokenizer_class": "BertTokenizer",
 
 
57
  "unk_token": "[UNK]"
58
  }
 
47
  "do_lower_case": false,
48
  "mask_token": "[MASK]",
49
  "max_len": 512,
50
+ "max_length": 512,
51
  "model_max_length": 512,
52
  "never_split": null,
53
+ "pad_to_multiple_of": null,
54
  "pad_token": "[PAD]",
55
+ "pad_token_type_id": 0,
56
+ "padding_side": "right",
57
  "sep_token": "[SEP]",
58
+ "stride": 0,
59
  "strip_accents": false,
60
  "tokenize_chinese_chars": true,
61
  "tokenizer_class": "BertTokenizer",
62
+ "truncation_side": "right",
63
+ "truncation_strategy": "longest_first",
64
  "unk_token": "[UNK]"
65
  }