kwang2049 commited on
Commit
fe50f47
·
1 Parent(s): 459c297
1_Pooling/config.json ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {
2
+ "word_embedding_dimension": 768,
3
+ "pooling_mode_cls_token": true,
4
+ "pooling_mode_mean_tokens": false,
5
+ "pooling_mode_max_tokens": false,
6
+ "pooling_mode_mean_sqrt_len_tokens": false
7
+ }
README.md ADDED
@@ -0,0 +1,122 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ pipeline_tag: sentence-similarity
3
+ tags:
4
+ - sentence-transformers
5
+ - feature-extraction
6
+ - sentence-similarity
7
+ - transformers
8
+ ---
9
+
10
+ # {MODEL_NAME}
11
+
12
+ This is a [sentence-transformers](https://www.SBERT.net) model: It maps sentences & paragraphs to a 768 dimensional dense vector space and can be used for tasks like clustering or semantic search.
13
+
14
+ <!--- Describe your model here -->
15
+
16
+ ## Usage (Sentence-Transformers)
17
+
18
+ Using this model becomes easy when you have [sentence-transformers](https://www.SBERT.net) installed:
19
+
20
+ ```
21
+ pip install -U sentence-transformers
22
+ ```
23
+
24
+ Then you can use the model like this:
25
+
26
+ ```python
27
+ from sentence_transformers import SentenceTransformer
28
+ sentences = ["This is an example sentence", "Each sentence is converted"]
29
+
30
+ model = SentenceTransformer('{MODEL_NAME}')
31
+ embeddings = model.encode(sentences)
32
+ print(embeddings)
33
+ ```
34
+
35
+
36
+
37
+ ## Usage (HuggingFace Transformers)
38
+ Without [sentence-transformers](https://www.SBERT.net), you can use the model like this: First, you pass your input through the transformer model, then you have to apply the right pooling-operation on-top of the contextualized word embeddings.
39
+
40
+ ```python
41
+ from transformers import AutoTokenizer, AutoModel
42
+ import torch
43
+
44
+
45
+ def cls_pooling(model_output, attention_mask):
46
+ return model_output[0][:,0]
47
+
48
+
49
+ # Sentences we want sentence embeddings for
50
+ sentences = ['This is an example sentence', 'Each sentence is converted']
51
+
52
+ # Load model from HuggingFace Hub
53
+ tokenizer = AutoTokenizer.from_pretrained('{MODEL_NAME}')
54
+ model = AutoModel.from_pretrained('{MODEL_NAME}')
55
+
56
+ # Tokenize sentences
57
+ encoded_input = tokenizer(sentences, padding=True, truncation=True, return_tensors='pt')
58
+
59
+ # Compute token embeddings
60
+ with torch.no_grad():
61
+ model_output = model(**encoded_input)
62
+
63
+ # Perform pooling. In this case, cls pooling.
64
+ sentence_embeddings = cls_pooling(model_output, encoded_input['attention_mask'])
65
+
66
+ print("Sentence embeddings:")
67
+ print(sentence_embeddings)
68
+ ```
69
+
70
+
71
+
72
+ ## Evaluation Results
73
+
74
+ <!--- Describe how your model was evaluated -->
75
+
76
+ For an automated evaluation of this model, see the *Sentence Embeddings Benchmark*: [https://seb.sbert.net](https://seb.sbert.net?model_name={MODEL_NAME})
77
+
78
+
79
+ ## Training
80
+ The model was trained with the parameters:
81
+
82
+ **DataLoader**:
83
+
84
+ `torch.utils.data.dataloader.DataLoader` of length 140000 with parameters:
85
+ ```
86
+ {'batch_size': 32, 'sampler': 'torch.utils.data.sampler.SequentialSampler', 'batch_sampler': 'torch.utils.data.sampler.BatchSampler'}
87
+ ```
88
+
89
+ **Loss**:
90
+
91
+ `gpl.toolkit.loss.MarginDistillationLoss`
92
+
93
+ Parameters of the fit()-Method:
94
+ ```
95
+ {
96
+ "epochs": 1,
97
+ "evaluation_steps": 0,
98
+ "evaluator": "NoneType",
99
+ "max_grad_norm": 1,
100
+ "optimizer_class": "<class 'transformers.optimization.AdamW'>",
101
+ "optimizer_params": {
102
+ "lr": 2e-05
103
+ },
104
+ "scheduler": "WarmupLinear",
105
+ "steps_per_epoch": 140000,
106
+ "warmup_steps": 1000,
107
+ "weight_decay": 0.01
108
+ }
109
+ ```
110
+
111
+
112
+ ## Full Model Architecture
113
+ ```
114
+ SentenceTransformer(
115
+ (0): Transformer({'max_seq_length': 350, 'do_lower_case': False}) with Transformer model: DistilBertModel
116
+ (1): Pooling({'word_embedding_dimension': 768, 'pooling_mode_cls_token': True, 'pooling_mode_mean_tokens': False, 'pooling_mode_max_tokens': False, 'pooling_mode_mean_sqrt_len_tokens': False})
117
+ )
118
+ ```
119
+
120
+ ## Citing & Authors
121
+
122
+ <!--- Describe where people can find more information -->
config.json ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_name_or_path": "/ukp-storage-1/kwang/.cache/torch/sentence_transformers/sentence-transformers_msmarco-distilbert-base-tas-b/",
3
+ "activation": "gelu",
4
+ "architectures": [
5
+ "DistilBertModel"
6
+ ],
7
+ "attention_dropout": 0.1,
8
+ "dim": 768,
9
+ "dropout": 0.1,
10
+ "hidden_dim": 3072,
11
+ "initializer_range": 0.02,
12
+ "max_position_embeddings": 512,
13
+ "model_type": "distilbert",
14
+ "n_heads": 12,
15
+ "n_layers": 6,
16
+ "pad_token_id": 0,
17
+ "qa_dropout": 0.1,
18
+ "seq_classif_dropout": 0.2,
19
+ "sinusoidal_pos_embds": false,
20
+ "tie_weights_": true,
21
+ "torch_dtype": "float32",
22
+ "transformers_version": "4.15.0",
23
+ "vocab_size": 30522
24
+ }
config_sentence_transformers.json ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {
2
+ "__version__": {
3
+ "sentence_transformers": "2.0.0",
4
+ "transformers": "4.7.0",
5
+ "pytorch": "1.9.0+cu102"
6
+ }
7
+ }
modules.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "idx": 0,
4
+ "name": "0",
5
+ "path": "",
6
+ "type": "sentence_transformers.models.Transformer"
7
+ },
8
+ {
9
+ "idx": 1,
10
+ "name": "1",
11
+ "path": "1_Pooling",
12
+ "type": "sentence_transformers.models.Pooling"
13
+ }
14
+ ]
pytorch_model.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d8e0fce1af744c7062eb557b072f2b70fc3928a7833582329e475b436258ccb2
3
+ size 265488185
results.json ADDED
@@ -0,0 +1,42 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "ndcg": {
3
+ "NDCG@1": 0.662,
4
+ "NDCG@3": 0.62948,
5
+ "NDCG@5": 0.62465,
6
+ "NDCG@10": 0.62367,
7
+ "NDCG@20": 0.64593,
8
+ "NDCG@100": 0.68044
9
+ },
10
+ "map": {
11
+ "MAP@1": 0.30429,
12
+ "MAP@3": 0.41961,
13
+ "MAP@5": 0.46933,
14
+ "MAP@10": 0.51537,
15
+ "MAP@20": 0.54365,
16
+ "MAP@100": 0.55674
17
+ },
18
+ "recall": {
19
+ "Recall@1": 0.30429,
20
+ "Recall@3": 0.45736,
21
+ "Recall@5": 0.53989,
22
+ "Recall@10": 0.63155,
23
+ "Recall@20": 0.71502,
24
+ "Recall@100": 0.82635
25
+ },
26
+ "precicion": {
27
+ "P@1": 0.662,
28
+ "P@3": 0.44867,
29
+ "P@5": 0.3624,
30
+ "P@10": 0.2462,
31
+ "P@20": 0.153,
32
+ "P@100": 0.03752
33
+ },
34
+ "mrr": {
35
+ "MRR@1": 0.664,
36
+ "MRR@3": 0.723,
37
+ "MRR@5": 0.7338,
38
+ "MRR@10": 0.73906,
39
+ "MRR@20": 0.74124,
40
+ "MRR@100": 0.74231
41
+ }
42
+ }
sentence_bert_config.json ADDED
@@ -0,0 +1,4 @@
 
 
 
 
 
1
+ {
2
+ "max_seq_length": 350,
3
+ "do_lower_case": false
4
+ }
special_tokens_map.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"unk_token": "[UNK]", "sep_token": "[SEP]", "pad_token": "[PAD]", "cls_token": "[CLS]", "mask_token": "[MASK]"}
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer_config.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"do_lower_case": true, "unk_token": "[UNK]", "sep_token": "[SEP]", "pad_token": "[PAD]", "cls_token": "[CLS]", "mask_token": "[MASK]", "tokenize_chinese_chars": true, "strip_accents": null, "do_basic_tokenize": true, "never_split": null, "model_max_length": 512, "name_or_path": "/ukp-storage-1/kwang/.cache/torch/sentence_transformers/sentence-transformers_msmarco-distilbert-base-tas-b/", "special_tokens_map_file": "/home/ukp-reimers/.cache/huggingface/transformers/ba1a276969ccad7ea2344196e7b8561b36292db74bff940ee316dadc05d005d3.dd8bd9bfd3664b530ea4e645105f557769387b3da9f79bdb55ed556bdd80611d", "tokenizer_class": "DistilBertTokenizer"}
vocab.txt ADDED
The diff for this file is too large to render. See raw diff