Initial commit

Browse files

Files changed (10) hide show

.gitattributes +1 -0
README.md +49 -0
added_tokens.json +1 -0
config.json +32 -0
entity_vocab.json +3 -0
pytorch_model.bin +3 -0
sentencepiece.bpe.model +3 -0
special_tokens_map.json +1 -0
tokenizer.json +0 -0
tokenizer_config.json +1 -0

.gitattributes CHANGED Viewed

@@ -31,3 +31,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text

 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
+entity_vocab.json filter=lfs diff=lfs merge=lfs -text

README.md CHANGED Viewed

@@ -1,3 +1,52 @@
 ---
 license: apache-2.0
 ---

 ---
+language: ja
+thumbnail: https://github.com/studio-ousia/luke/raw/master/resources/luke_logo.png
+tags:
+  - luke
+  - named entity recognition
+  - entity typing
+  - relation classification
+  - question answering
 license: apache-2.0
 ---
+## luke-japanese
+**luke-japanese** is the Japanese version of **LUKE** (**L**anguage
+**U**nderstanding with **K**nowledge-based **E**mbeddings), a pre-trained
+_knowledge-enhanced_ contextualized representation of words and entities based
+on transformer. LUKE treats words and entities in a given text as independent
+tokens, and outputs contextualized representations of them. Please refer to our
+[GitHub repository](https://github.com/studio-ousia/luke) for more details and
+updates.
+**luke-japanese**は、単語とエンティティの知識拡張型訓練済みモデル**LUKE**の日本
+語版です。LUKE は単語とエンティティを独立したトークンとして扱い、これらの文脈を
+考慮した表現を出力します。詳細については
+、[GitHub リポジトリ](https://github.com/studio-ousia/luke)を参照してください。
+### Experimental results on JGLUE
+The performance of luke-japanese evaluated on the dev set of
+[JGLUE](https://github.com/yahoojapan/JGLUE) is shown as follows:
+| Model                  | MARC-ja   | JSTS                | JNLI      | JCommonsenseQA |
+| ---------------------- | --------- | ------------------- | --------- | -------------- |
+|                        | acc       | Pearson/Spearman    | acc       | acc            |
+| **luke-japanese-base** | **0.963** | **0.912**/**0.875** | **0.912** | **0.842**      |
+| _Baselines:_           |           |
+| Tohoku BERT base       | 0.958     | 0.899/0.859         | 0.899     | 0.808          |
+| NICT BERT base         | 0.958     | 0.903/0.867         | 0.902     | 0.823          |
+| Waseda RoBERTa base    | 0.962     | 0.901/0.865         | 0.895     | 0.840          |
+| XLM RoBERTa base       | 0.961     | 0.870/0.825         | 0.893     | 0.687          |
+### Citation
+```latex
+@inproceedings{yamada2020luke,
+  title={LUKE: Deep Contextualized Entity Representations with Entity-aware Self-attention},
+  author={Ikuya Yamada and Akari Asai and Hiroyuki Shindo and Hideaki Takeda and Yuji Matsumoto},
+  booktitle={EMNLP},
+  year={2020}
+}
+```

added_tokens.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"<ent>": 32770, "<ent2>": 32771}

config.json ADDED Viewed

	@@ -0,0 +1,32 @@

+{
+  "_name_or_path": "models/luke-japanese/hf_xlm_roberta",
+  "architectures": [
+    "LukeForMaskedLM"
+  ],
+  "attention_probs_dropout_prob": 0.1,
+  "bert_model_name": "models/luke-japanese/hf_xlm_roberta",
+  "bos_token_id": 0,
+  "classifier_dropout": null,
+  "cls_entity_prediction": false,
+  "entity_emb_size": 256,
+  "entity_vocab_size": 570505,
+  "eos_token_id": 2,
+  "hidden_act": "gelu",
+  "hidden_dropout_prob": 0.1,
+  "hidden_size": 768,
+  "initializer_range": 0.02,
+  "intermediate_size": 3072,
+  "layer_norm_eps": 1e-05,
+  "max_position_embeddings": 514,
+  "model_type": "luke",
+  "num_attention_heads": 12,
+  "num_hidden_layers": 12,
+  "pad_token_id": 1,
+  "position_embedding_type": "absolute",
+  "torch_dtype": "float32",
+  "transformers_version": "4.13.0",
+  "type_vocab_size": 1,
+  "use_cache": true,
+  "use_entity_aware_attention": true,
+  "vocab_size": 32772
+}

entity_vocab.json ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:a7b569f330b5ddbeae34dee4ac4d4681585f2b6358cffbb372829233be1606aa
+size 20543383

pytorch_model.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:f40c0932218c73f6c167c76b84e1e35979dc2cd93f8f628f8c8e7ee8a05ebf59
+size 1122140419

sentencepiece.bpe.model ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:d8b73a5e054936c920cf5b7d1ec21ce9c281977078269963beb821c6c86fbff7
+size 841889

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1 @@

+ {"bos_token": "<s>", "eos_token": "</s>", "unk_token": "<unk>", "sep_token": "</s>", "pad_token": "<pad>", "cls_token": "<s>", "mask_token": {"content": "<mask>", "single_word": false, "lstrip": true, "rstrip": false, "normalized": true}, "additional_special_tokens": [{"content": "<ent>", "single_word": false, "lstrip": false, "rstrip": false, "normalized": false}, {"content": "<ent2>", "single_word": false, "lstrip": false, "rstrip": false, "normalized": false}]}

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1 @@

+ {"bos_token": "<s>", "eos_token": "</s>", "unk_token": "<unk>", "sep_token": "</s>", "cls_token": "<s>", "pad_token": "<pad>", "mask_token": {"content": "<mask>", "single_word": false, "lstrip": true, "rstrip": false, "normalized": true, "__type": "AddedToken"}, "sp_model_kwargs": {}, "task": null, "max_entity_length": 32, "max_mention_length": 30, "entity_token_1": {"content": "<ent>", "single_word": false, "lstrip": false, "rstrip": false, "normalized": true, "__type": "AddedToken"}, "entity_token_2": {"content": "<ent2>", "single_word": false, "lstrip": false, "rstrip": false, "normalized": true, "__type": "AddedToken"}, "model_max_length": 512, "special_tokens_map_file": "models/luke-japanese/hf_xlm_roberta/special_tokens_map.json", "name_or_path": "models/luke-japanese/hf_luke_japanese_epoch20", "tokenizer_file": "models/luke-japanese/hf_luke_japanese_epoch20/tokenizer.json", "additional_special_tokens": [{"content": "<ent>", "single_word": false, "lstrip": false, "rstrip": false, "normalized": true, "__type": "AddedToken"}, {"content": "<ent2>", "single_word": false, "lstrip": false, "rstrip": false, "normalized": true, "__type": "AddedToken"}], "tokenizer_class": "MLukeTokenizer"}