ner / metadata.json
DoppelApp's picture
runtime contract only
fa88510 verified
Raw
History Blame Contribute Delete
5.66 kB
{
"format_version": 1,
"artifacts": {
"mlpackage": "GLiNERRelexBase.mlpackage",
"mlmodelc": "GLiNERRelexBase.mlmodelc"
},
"tokenizer": {
"type": "DebertaV2 (SentencePiece unigram), is_split_into_words semantics",
"files": {
"tokenizer_json": "tokenizer.json",
"sentencepiece_model": "spm.model",
"tokenizer_config": "tokenizer_config.json",
"added_tokens": "added_tokens.json",
"special_tokens_map": "special_tokens_map.json"
},
"special_token_ids": {
"pad": 0,
"cls": 1,
"sep": 2,
"unk": 3,
"mask": 128000,
"ent_prompt": 128001,
"sep_prompt": 128002,
"rel_prompt": 128003
}
},
"prompt_construction": {
"description": "Word-level prompt prepended to the whitespace-split text words, then tokenized as pre-split words ([CLS] + subtokens + [SEP], each word SentencePiece-encoded with a leading \u2581).",
"word_layout": "[<<ENT>>, label_1, <<ENT>>, label_2, ..., <<SEP>>, <<REL>>, relation_1, <<REL>>, relation_2, ..., <<SEP>>, text_word_1, ..., text_word_L]",
"words_splitter_regex": "\\w+(?:[-_]\\w+)*|\\S",
"notes": [
"Both <<SEP>> tokens are required: one closes the entity-label block, one closes the relation block.",
"Labels/relations must be deduplicated preserving order (dict.fromkeys semantics).",
"The k-th <<ENT>> token feeds entity-class slot k-1; the k-th <<REL>> token feeds relation-class slot k-1."
]
},
"inputs": {
"input_ids": {
"shape": [
1,
512
],
"dtype": "int32",
"semantics": "DeBERTa subtoken ids, [PAD]=0 right-padded"
},
"attention_mask": {
"shape": [
1,
512
],
"dtype": "int32",
"semantics": "1 for real subtokens (incl. CLS/SEP), 0 for padding"
},
"words_mask": {
"shape": [
1,
512
],
"dtype": "int32",
"semantics": "1-based text-word index on the FIRST subtoken of each text word; 0 on prompt words, special tokens, continuation subtokens and padding"
},
"text_lengths": {
"shape": [
1,
1
],
"dtype": "int32",
"semantics": "number of text words L (excludes prompt words)"
},
"span_idx": {
"shape": [
1,
16,
2
],
"dtype": "int32",
"semantics": "word-level [start, end] (inclusive) of extracted candidate entities for the relation pass; zeros elsewhere"
},
"span_mask": {
"shape": [
1,
16
],
"dtype": "int32",
"semantics": "1 for occupied entity slots, else 0"
}
},
"outputs": {
"logits": {
"shape": [
1,
320,
12,
3
],
"dtype": "float32",
"axes": [
"batch",
"word position (only w < text_lengths valid)",
"entity class slot (only c < num_labels valid)",
"BIO component: 0=start, 1=end, 2=inside"
],
"semantics": "raw token BIO logits; apply sigmoid before thresholding"
},
"rel_logits": {
"shape": [
1,
240,
8
],
"dtype": "float32",
"axes": [
"batch",
"ordered entity-slot pair",
"relation class slot (only r < num_relations valid)"
],
"pair_slot_formula": "slot(i, j) = i * (MAX_ENT_SLOTS - 1) + (j if j < i else j - 1), i != j",
"semantics": "raw relation logits for pair (head=slot i, tail=slot j); apply sigmoid; only pairs with both slots valid in span_mask are meaningful"
}
},
"static_shapes": {
"seq_len": 512,
"max_words": 320,
"max_entity_types": 12,
"max_relation_types": 8,
"max_entity_slots": 16,
"num_pairs": 240,
"hidden_size": 768
},
"decode_algorithm": {
"protocol": "two-pass: pass1 span_idx/span_mask all zero -> logits; extract candidate spans; pass2 with spans -> rel_logits",
"entity_threshold": 0.4,
"relation_threshold": 0.7,
"flat_ner": false,
"multi_label": false,
"candidate_span_extraction": "For every (word, class) with sigmoid(start) > t and every (word', class) with sigmoid(end) > t, word <= word', keep (word, word') if sigmoid(inside) > t for the whole run. Iterate starts row-major (word-major) then ends; keep duplicates from different classes. These ordered spans fill span_idx slots 0..E-1.",
"entity_decoding": "Same start/end/inside pairing on sigmoid probabilities; span score = min(start, end, all inside); then greedy overlap filtering: sort by score desc (stable), keep spans that do not conflict (flat_ner=false allows nesting, identical boundaries conflict unless multi_label); finally sort by start (stable).",
"relation_decoding": "Map each candidate slot to the FIRST decoded entity with identical word boundaries (slots without a match are skipped). For each ordered pair (i, j), i != j, over slots 0..E-1 in row-major order, emit (head, relation_r, tail, sigmoid) for every relation class r with sigmoid(rel_logits[slot(i,j), r]) > relation_threshold. Duplicate-boundary slots intentionally emit duplicate relations (PyTorch decoder behavior)."
},
"limits": {
"max_text_words": 320,
"max_subtokens": 512,
"max_labels": 12,
"max_relations": 8,
"max_candidate_spans": 16,
"unused_class_columns": "logit columns for class slots beyond the provided label count contain garbage (zero prompt embeddings) and MUST be ignored"
},
"sha256": {
"mlpackage_tree": "9d948eb9e9c690ee5941ab28261361dad00ce74d70f997ca35ac9e09a6e4ab0a",
"mlmodelc_tree": "f700158783796a7b93e366708c8467899ac6d4c7e18da567537046e1dfa43a5f"
}
}