Token Classification
GLiNER
PyTorch
English
entity recognition
named-entity-recognition
zero-shot
zero-shot-ner
zero shot
biomedical-nlp
cancer-genetics
oncology
gene-regulation
cancer-research
amino_acid
anatomical_system
cancer
cell
cellular_component
developing_anatomical_structure
gene_or_gene_product
immaterial_anatomical_entity
multi-tissue_structure
organ
organism
organism_subdivision
organism_substance
pathological_formation
simple_chemical
tissue
Instructions to use OpenMed/OpenMed-ZeroShot-NER-Oncology-Base-220M with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- GLiNER
How to use OpenMed/OpenMed-ZeroShot-NER-Oncology-Base-220M with GLiNER:
from gliner import GLiNER model = GLiNER.from_pretrained("OpenMed/OpenMed-ZeroShot-NER-Oncology-Base-220M") - Notebooks
- Google Colab
- Kaggle
File size: 3,528 Bytes
0f160e2 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 | {
"class_token_index": 250100,
"dropout": 0.3,
"embed_ent_token": true,
"encoder_config": {
"_name_or_path": "google/mt5-base",
"add_cross_attention": false,
"architectures": [
"MT5ForConditionalGeneration"
],
"bad_words_ids": null,
"begin_suppress_tokens": null,
"bos_token_id": null,
"chunk_size_feed_forward": 0,
"classifier_dropout": 0.0,
"cross_attention_hidden_size": null,
"d_ff": 2048,
"d_kv": 64,
"d_model": 768,
"decoder_start_token_id": 0,
"dense_act_fn": "gelu_new",
"diversity_penalty": 0.0,
"do_sample": false,
"dropout_rate": 0.1,
"early_stopping": false,
"encoder_no_repeat_ngram_size": 0,
"eos_token_id": 1,
"exponential_decay_length_penalty": null,
"feed_forward_proj": "gated-gelu",
"finetuning_task": null,
"forced_bos_token_id": null,
"forced_eos_token_id": null,
"id2label": {
"0": "LABEL_0",
"1": "LABEL_1"
},
"initializer_factor": 1.0,
"is_decoder": false,
"is_encoder_decoder": true,
"is_gated_act": true,
"label2id": {
"LABEL_0": 0,
"LABEL_1": 1
},
"layer_norm_epsilon": 1e-06,
"length_penalty": 1.0,
"max_length": 20,
"min_length": 0,
"model_type": "mt5",
"no_repeat_ngram_size": 0,
"num_beam_groups": 1,
"num_beams": 1,
"num_decoder_layers": 12,
"num_heads": 12,
"num_layers": 12,
"num_return_sequences": 1,
"output_attentions": false,
"output_hidden_states": false,
"output_past": true,
"output_scores": false,
"pad_token_id": 0,
"prefix": null,
"problem_type": null,
"pruned_heads": {},
"relative_attention_max_distance": 128,
"relative_attention_num_buckets": 32,
"remove_invalid_values": false,
"repetition_penalty": 1.0,
"return_dict": true,
"return_dict_in_generate": false,
"sep_token_id": null,
"suppress_tokens": null,
"task_specific_params": null,
"temperature": 1.0,
"tf_legacy_loss": false,
"tie_encoder_decoder": false,
"tie_word_embeddings": false,
"tokenizer_class": "T5Tokenizer",
"top_k": 50,
"top_p": 1.0,
"torch_dtype": null,
"torchscript": false,
"typical_p": 1.0,
"use_bfloat16": false,
"use_cache": true,
"vocab_size": 250102
},
"ent_token": "<<ENT>>",
"eval_every": 10000,
"fine_tune": true,
"freeze_token_rep": false,
"fuse_layers": false,
"has_rnn": true,
"hidden_size": 768,
"label_smoothing": 0,
"labels_encoder": null,
"labels_encoder_config": null,
"log_dir": "models/",
"loss_alpha": 0.75,
"loss_gamma": 0,
"loss_reduction": "sum",
"lr_encoder": "1e-5",
"lr_others": "3e-5",
"max_grad_norm": 10.0,
"max_len": 1024,
"max_neg_type_ratio": 1,
"max_types": 30,
"max_width": 12,
"model_name": "google/mt5-base",
"model_type": "gliner",
"name": "span level gliner",
"num_post_fusion_layers": 1,
"num_steps": 80000,
"post_fusion_schema": "",
"prev_path": null,
"random_drop": true,
"root_dir": "gliner_logs",
"save_total_limit": 3,
"scheduler_type": "cosine",
"sep_token": "<<SEP>>",
"shuffle_types": true,
"size_sup": -1,
"span_mode": "markerV0",
"subtoken_pooling": "first",
"train_batch_size": 8,
"train_data": "data/multilingual_data.json",
"transformers_version": "4.43.4",
"val_data_dir": "none",
"vocab_size": 250102,
"warmup_ratio": 0.05,
"weight_decay_encoder": 0.1,
"weight_decay_other": 0.01,
"words_splitter_type": "universal"
}
|