VRDate's picture
Set up static file serving for schema:
9d094e3
Raw
History Blame Contribute Delete
11.6 kB
{
"$schema": "http://json-schema.org/draft-07/schema#",
"$id": "https://github.com/hebrew-unified-nlp/schema/v2.1.0",
"title": "HebrewUnifiedNLPResult",
"description": "Async parallel Hebrew/English NLP with cross-platform sentence breaking - outputs sentence array with per-sentence language detection",
"type": "object",
"required": [
"meta",
"input",
"sentences"
],
"properties": {
"meta": {
"type": "object",
"description": "Processing metadata",
"required": [
"version",
"timestamp",
"sentence_count"
],
"properties": {
"version": {
"type": "string",
"pattern": "^\\d+\\.\\d+\\.\\d+$",
"example": "2025.12.5"
},
"timestamp": {
"type": "string",
"format": "date-time"
},
"models": {
"type": "object",
"properties": {
"phonikud": {
"type": "string",
"description": "Phonikud model name"
},
"piper": {
"type": "string",
"description": "Piper voice manager info"
},
"dictabert": {
"type": "string",
"description": "DictaBERT model path"
}
}
},
"processing_time_ms": {
"type": "number",
"minimum": 0,
"description": "Total processing time for all sentences"
},
"sentence_count": {
"type": "integer",
"minimum": 0,
"description": "Number of sentences detected"
},
"parallel_workers": {
"type": "integer",
"minimum": 1,
"description": "Number of parallel workers used"
},
"sentence_breaker": {
"type": "string",
"description": "Sentence breaking backend used",
"enum": [
"pysbd",
"pysbd-en",
"icu",
"wtpsplit",
"regex"
]
}
}
},
"translations": {
"type": "object",
"description": "Bilingual label translations (Hebrew/English)",
"properties": {
"pos": {
"type": "object",
"description": "Part-of-speech translations",
"additionalProperties": {
"$ref": "#/definitions/BilingualLabel"
}
},
"dep": {
"type": "object",
"description": "Dependency relation translations",
"additionalProperties": {
"$ref": "#/definitions/BilingualLabel"
}
},
"ner": {
"type": "object",
"description": "Named entity type translations",
"additionalProperties": {
"$ref": "#/definitions/BilingualLabel"
}
},
"prefix": {
"type": "object",
"description": "Prefix type translations",
"additionalProperties": {
"$ref": "#/definitions/BilingualLabel"
}
},
"special": {
"type": "object",
"description": "Special token translations",
"additionalProperties": {
"$ref": "#/definitions/BilingualLabel"
}
},
"morph": {
"type": "object",
"description": "Morphological feature value translations",
"additionalProperties": {
"$ref": "#/definitions/BilingualLabel"
}
}
}
},
"input": {
"type": "object",
"required": [
"text"
],
"properties": {
"text": {
"type": "string",
"description": "Original input text (may contain multiple sentences)"
},
"language": {
"type": "string",
"description": "Primary input language",
"enum": [
"he",
"en"
]
}
}
},
"sentences": {
"type": "array",
"description": "Array of sentence analysis results (parallel processed)",
"items": {
"$ref": "#/definitions/SentenceResult"
}
}
},
"definitions": {
"BilingualLabel": {
"type": "object",
"description": "Label with Hebrew and English translations",
"properties": {
"en": {
"type": "string"
},
"he": {
"type": "string"
}
},
"required": [
"en",
"he"
]
},
"SentenceResult": {
"type": "object",
"description": "Analysis result for a single sentence",
"required": [
"index",
"lang",
"text",
"phonetics",
"tokens"
],
"properties": {
"index": {
"type": "integer",
"minimum": 0,
"description": "Sentence index in original text"
},
"lang": {
"type": "string",
"enum": [
"he",
"en"
],
"description": "Detected language for this sentence"
},
"text": {
"type": "string",
"description": "Original sentence text"
},
"phonetics": {
"$ref": "#/definitions/Phonetics"
},
"tokens": {
"type": "array",
"items": {
"$ref": "#/definitions/Token"
}
},
"ner_entities": {
"type": "array",
"items": {
"$ref": "#/definitions/NEREntity"
}
},
"speech": {
"oneOf": [
{
"$ref": "#/definitions/Speech"
},
{
"type": "null"
}
]
},
"tree_svg": {
"type": "string",
"description": "SVG string of dependency tree visualization"
}
}
},
"Phonetics": {
"type": "object",
"description": "Phonikud output for sentence (Hebrew) or passthrough (English)",
"required": [
"diacritized",
"phonemes"
],
"properties": {
"diacritized": {
"type": "string",
"description": "Sentence with nikud (Hebrew) or original text (English)"
},
"phonemes": {
"type": "string",
"description": "Phoneme sequence (Hebrew) or empty string (English)"
}
}
},
"Token": {
"type": "object",
"description": "Unified token: DictaBERT + Phonikud enrichment",
"required": [
"id",
"token"
],
"properties": {
"id": {
"type": "integer",
"minimum": 1,
"description": "1-indexed token position within sentence"
},
"token": {
"type": "string",
"description": "Surface form (DictaBERT)"
},
"nikud": {
"type": "string",
"description": "Diacritized form (Phonikud for Hebrew, original for English)"
},
"phonemes": {
"type": "string",
"description": "Token phonemes (Hebrew) or empty (English)"
},
"seg": {
"type": "array",
"description": "Morphological segmentation (DictaBERT)",
"items": {
"type": "string"
}
},
"lex": {
"type": "string",
"description": "Lemma (DictaBERT)"
},
"offsets": {
"type": "object",
"properties": {
"start": {
"type": "integer"
},
"end": {
"type": "integer"
}
}
},
"morph": {
"$ref": "#/definitions/Morphology"
},
"syntax": {
"$ref": "#/definitions/Syntax"
}
}
},
"Morphology": {
"type": "object",
"description": "DictaBERT morphological analysis",
"properties": {
"token": {
"type": "string"
},
"pos": {
"type": "string",
"enum": [
"ADJ",
"ADP",
"ADV",
"AUX",
"CCONJ",
"DET",
"INTJ",
"NOUN",
"NUM",
"PART",
"PRON",
"PROPN",
"PUNCT",
"SCONJ",
"SYM",
"VERB",
"X"
]
},
"feats": {
"type": "object",
"properties": {
"Gender": {
"type": "string",
"enum": [
"Masc",
"Fem"
]
},
"Number": {
"type": "string",
"enum": [
"Sing",
"Plur",
"Dual"
]
},
"Person": {
"type": "string",
"enum": [
"1",
"2",
"3"
]
},
"Tense": {
"type": "string",
"enum": [
"Past",
"Present",
"Future",
"Imp",
"Inf"
]
},
"Voice": {
"type": "string",
"enum": [
"Act",
"Pass"
]
},
"Definite": {
"type": "string",
"enum": [
"Def",
"Ind"
]
},
"Case": {
"type": "string",
"enum": [
"Nom",
"Acc",
"Gen"
]
},
"Construct": {
"type": "string",
"enum": [
"Construct",
"Free"
]
}
},
"additionalProperties": true
},
"prefixes": {
"type": "array",
"items": {
"type": "string"
}
},
"suffix": {
"type": "boolean"
}
}
},
"Syntax": {
"type": "object",
"description": "DictaBERT dependency syntax",
"properties": {
"word": {
"type": "string"
},
"dep_head_idx": {
"type": "integer",
"minimum": -1,
"description": "-1 for root"
},
"dep_func": {
"type": "string"
},
"dep_head": {
"type": "string"
}
}
},
"NEREntity": {
"type": "object",
"properties": {
"phrase": {
"type": "string",
"description": "Entity text"
},
"label": {
"type": "string",
"enum": [
"PER",
"LOC",
"GPE",
"ORG",
"TIME",
"DATE",
"MISC",
"MONEY",
"PERCENT"
]
},
"start": {
"type": "integer"
},
"end": {
"type": "integer"
}
}
},
"Speech": {
"type": "object",
"description": "TTS audio for sentence",
"properties": {
"format": {
"type": "string",
"enum": [
"wav",
"mp3",
"ogg"
]
},
"sample_rate": {
"type": "integer"
},
"duration_ms": {
"type": "number"
},
"data_uri": {
"type": "string",
"description": "Base64 data URI (data:audio/wav;base64,...)"
}
}
}
}
}