lfoppiano commited on
Commit
3dc4a1e
·
1 Parent(s): ab9222f

Add delft 0.4.6 load scaffold for material-BERT_CRF (preprocessor.json + transformer config/tokenizer; no retrain) (#2)

Browse files

- Add delft 0.4.6 load scaffold for material-BERT_CRF (preprocessor.json + transformer config/tokenizer; no retrain) (4f7a5e76b663b2c3cc1fa5ca399975ce56e70e84)

material-BERT_CRF/config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model_name": "material-BERT_CRF",
3
+ "architecture": "BERT_CRF",
4
+ "embeddings_name": null,
5
+ "char_vocab_size": 123,
6
+ "case_vocab_size": 8,
7
+ "char_embedding_size": 25,
8
+ "num_char_lstm_units": 25,
9
+ "max_char_length": 30,
10
+ "features_vocabulary_size": 12,
11
+ "features_indices": null,
12
+ "features_embedding_size": 4,
13
+ "features_lstm_units": 4,
14
+ "max_sequence_length": 512,
15
+ "word_embedding_size": 0,
16
+ "num_word_lstm_units": 100,
17
+ "case_embedding_size": 5,
18
+ "dropout": 0.5,
19
+ "recurrent_dropout": 0.5,
20
+ "use_crf": true,
21
+ "use_chain_crf": false,
22
+ "fold_number": 1,
23
+ "batch_size": 30,
24
+ "transformer_name": "allenai/scibert_scivocab_cased/dir",
25
+ "use_ELMo": false,
26
+ "labels": {
27
+ "<PAD>": 0,
28
+ "B-<doping>": 1,
29
+ "B-<fabrication>": 2,
30
+ "B-<formula>": 3,
31
+ "B-<name>": 4,
32
+ "B-<shape>": 5,
33
+ "B-<substrate>": 6,
34
+ "B-<value>": 7,
35
+ "B-<variable>": 8,
36
+ "I-<doping>": 9,
37
+ "I-<fabrication>": 10,
38
+ "I-<formula>": 11,
39
+ "I-<name>": 12,
40
+ "I-<shape>": 13,
41
+ "I-<substrate>": 14,
42
+ "I-<value>": 15,
43
+ "I-<variable>": 16,
44
+ "O": 17
45
+ }
46
+ }
material-BERT_CRF/preprocessor.json ADDED
@@ -0,0 +1,65 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "padding": true,
3
+ "return_lengths": true,
4
+ "return_word_embeddings": false,
5
+ "return_casing": false,
6
+ "return_features": false,
7
+ "return_chars": false,
8
+ "return_bert_embeddings": true,
9
+ "vocab_char": {
10
+ "<PAD>": 0,
11
+ "<UNK>": 1
12
+ },
13
+ "vocab_tag": {
14
+ "<PAD>": 0,
15
+ "B-<doping>": 1,
16
+ "B-<fabrication>": 2,
17
+ "B-<formula>": 3,
18
+ "B-<name>": 4,
19
+ "B-<shape>": 5,
20
+ "B-<substrate>": 6,
21
+ "B-<value>": 7,
22
+ "B-<variable>": 8,
23
+ "I-<doping>": 9,
24
+ "I-<fabrication>": 10,
25
+ "I-<formula>": 11,
26
+ "I-<name>": 12,
27
+ "I-<shape>": 13,
28
+ "I-<substrate>": 14,
29
+ "I-<value>": 15,
30
+ "I-<variable>": 16,
31
+ "O": 17
32
+ },
33
+ "vocab_case": [
34
+ "<PAD>",
35
+ "numeric",
36
+ "allLower",
37
+ "allUpper",
38
+ "initialUpper",
39
+ "other",
40
+ "mainly_numeric",
41
+ "contains_digit"
42
+ ],
43
+ "max_char_length": 30,
44
+ "feature_preprocessor": null,
45
+ "indice_tag": {
46
+ "0": "<PAD>",
47
+ "1": "B-<doping>",
48
+ "2": "B-<fabrication>",
49
+ "3": "B-<formula>",
50
+ "4": "B-<name>",
51
+ "5": "B-<shape>",
52
+ "6": "B-<substrate>",
53
+ "7": "B-<value>",
54
+ "8": "B-<variable>",
55
+ "9": "I-<doping>",
56
+ "10": "I-<fabrication>",
57
+ "11": "I-<formula>",
58
+ "12": "I-<name>",
59
+ "13": "I-<shape>",
60
+ "14": "I-<substrate>",
61
+ "15": "I-<value>",
62
+ "16": "I-<variable>",
63
+ "17": "O"
64
+ }
65
+ }
material-BERT_CRF/transformer-config.json ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_name_or_path": "allenai/scibert_scivocab_cased",
3
+ "attention_probs_dropout_prob": 0.1,
4
+ "classifier_dropout": null,
5
+ "hidden_act": "gelu",
6
+ "hidden_dropout_prob": 0.1,
7
+ "hidden_size": 768,
8
+ "initializer_range": 0.02,
9
+ "intermediate_size": 3072,
10
+ "layer_norm_eps": 1e-12,
11
+ "max_position_embeddings": 512,
12
+ "model_type": "bert",
13
+ "num_attention_heads": 12,
14
+ "num_hidden_layers": 12,
15
+ "pad_token_id": 0,
16
+ "position_embedding_type": "absolute",
17
+ "transformers_version": "4.48.0",
18
+ "type_vocab_size": 2,
19
+ "use_cache": true,
20
+ "vocab_size": 31116
21
+ }
material-BERT_CRF/transformer-tokenizer/special_tokens_map.json ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {
2
+ "cls_token": "[CLS]",
3
+ "mask_token": "[MASK]",
4
+ "pad_token": "[PAD]",
5
+ "sep_token": "[SEP]",
6
+ "unk_token": "[UNK]"
7
+ }
material-BERT_CRF/transformer-tokenizer/tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
material-BERT_CRF/transformer-tokenizer/tokenizer_config.json ADDED
@@ -0,0 +1,58 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "added_tokens_decoder": {
3
+ "0": {
4
+ "content": "[PAD]",
5
+ "lstrip": false,
6
+ "normalized": false,
7
+ "rstrip": false,
8
+ "single_word": false,
9
+ "special": true
10
+ },
11
+ "100": {
12
+ "content": "[UNK]",
13
+ "lstrip": false,
14
+ "normalized": false,
15
+ "rstrip": false,
16
+ "single_word": false,
17
+ "special": true
18
+ },
19
+ "101": {
20
+ "content": "[CLS]",
21
+ "lstrip": false,
22
+ "normalized": false,
23
+ "rstrip": false,
24
+ "single_word": false,
25
+ "special": true
26
+ },
27
+ "102": {
28
+ "content": "[SEP]",
29
+ "lstrip": false,
30
+ "normalized": false,
31
+ "rstrip": false,
32
+ "single_word": false,
33
+ "special": true
34
+ },
35
+ "103": {
36
+ "content": "[MASK]",
37
+ "lstrip": false,
38
+ "normalized": false,
39
+ "rstrip": false,
40
+ "single_word": false,
41
+ "special": true
42
+ }
43
+ },
44
+ "clean_up_tokenization_spaces": true,
45
+ "cls_token": "[CLS]",
46
+ "do_basic_tokenize": true,
47
+ "do_lower_case": true,
48
+ "extra_special_tokens": {},
49
+ "mask_token": "[MASK]",
50
+ "model_max_length": 1000000000000000019884624838656,
51
+ "never_split": null,
52
+ "pad_token": "[PAD]",
53
+ "sep_token": "[SEP]",
54
+ "strip_accents": null,
55
+ "tokenize_chinese_chars": true,
56
+ "tokenizer_class": "BertTokenizer",
57
+ "unk_token": "[UNK]"
58
+ }
material-BERT_CRF/transformer-tokenizer/vocab.txt ADDED
The diff for this file is too large to render. See raw diff