96abhishekarora's picture
Updated model with better training and evaluation. Test and val data included as pickle files. Older Legacy files were removed to avoid confusion.
2de46f4
raw
history blame contribute delete
874 Bytes
{
"_name_or_path": "oshizo/sbert-jsnli-luke-japanese-base-lite",
"architectures": [
"LukeModel"
],
"attention_probs_dropout_prob": 0.1,
"bert_model_name": "models/luke-japanese/hf_xlm_roberta",
"bos_token_id": 0,
"classifier_dropout": null,
"cls_entity_prediction": false,
"entity_emb_size": 256,
"entity_vocab_size": 4,
"eos_token_id": 2,
"hidden_act": "gelu",
"hidden_dropout_prob": 0.1,
"hidden_size": 768,
"initializer_range": 0.02,
"intermediate_size": 3072,
"layer_norm_eps": 1e-05,
"max_position_embeddings": 514,
"model_type": "luke",
"num_attention_heads": 12,
"num_hidden_layers": 12,
"pad_token_id": 1,
"position_embedding_type": "absolute",
"torch_dtype": "float32",
"transformers_version": "4.35.1",
"type_vocab_size": 1,
"use_cache": true,
"use_entity_aware_attention": true,
"vocab_size": 32772
}