Clarify JEVision model identity in visual examples and API
Browse files- README.md +2 -0
- examples/long-context/input.json +1 -1
- examples/real-photo/input.json +1 -1
- examples/real-photo/noul-input.json +1 -1
- examples/real-photo/output.json +1 -1
- runtime/kev/api.py +1 -1
- runtime/kev/jevvision.py +1 -1
- runtime/kev/serve.py +4 -3
README.md
CHANGED
|
@@ -134,6 +134,8 @@ The saved [request](examples/real-photo/input.json) and [full response](examples
|
|
| 134 |
}
|
| 135 |
```
|
| 136 |
|
|
|
|
|
|
|
| 137 |
This is one recorded example. The full response also records the latency of that CPU run; it is not a speed comparison. Run it yourself after starting the server:
|
| 138 |
|
| 139 |
```bash
|
|
|
|
| 134 |
}
|
| 135 |
```
|
| 136 |
|
| 137 |
+
In these examples, `model: "jevvision"` is the API label echoed in the response; it does not choose the inference weights. The `images` field routes the request through the JEVision visual adapter and pointer head in this repository. The optional KEV-0.8B checkpoint is used only when selected for text-only requests. The saved photo response retains its recorded answer, probabilities, token counts, and latency; only its echoed model label was normalized from the older API alias.
|
| 138 |
+
|
| 139 |
This is one recorded example. The full response also records the latency of that CPU run; it is not a speed comparison. Run it yourself after starting the server:
|
| 140 |
|
| 141 |
```bash
|
examples/long-context/input.json
CHANGED
|
@@ -1,5 +1,5 @@
|
|
| 1 |
{
|
| 2 |
-
"model": "
|
| 3 |
"state_target_tokens": 76000,
|
| 4 |
"minimum_request_tokens": 75000,
|
| 5 |
"state_file": "state-75k.txt",
|
|
|
|
| 1 |
{
|
| 2 |
+
"model": "jevvision",
|
| 3 |
"state_target_tokens": 76000,
|
| 4 |
"minimum_request_tokens": 75000,
|
| 5 |
"state_file": "state-75k.txt",
|
examples/real-photo/input.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
| 1 |
{
|
| 2 |
"state": "Use the attached photo as visual context.",
|
| 3 |
-
"model": "
|
| 4 |
"image_file": "coffee-and-laptop.jpg",
|
| 5 |
"questions": {
|
| 6 |
"scene": {
|
|
|
|
| 1 |
{
|
| 2 |
"state": "Use the attached photo as visual context.",
|
| 3 |
+
"model": "jevvision",
|
| 4 |
"image_file": "coffee-and-laptop.jpg",
|
| 5 |
"questions": {
|
| 6 |
"scene": {
|
examples/real-photo/noul-input.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
| 1 |
{
|
| 2 |
"state": "Use the attached photo as visual context.",
|
| 3 |
-
"model": "
|
| 4 |
"image_file": "coffee-and-laptop.jpg",
|
| 5 |
"questions": {
|
| 6 |
"is_macbook": {
|
|
|
|
| 1 |
{
|
| 2 |
"state": "Use the attached photo as visual context.",
|
| 3 |
+
"model": "jevvision",
|
| 4 |
"image_file": "coffee-and-laptop.jpg",
|
| 5 |
"questions": {
|
| 6 |
"is_macbook": {
|
examples/real-photo/output.json
CHANGED
|
@@ -1,5 +1,5 @@
|
|
| 1 |
{
|
| 2 |
-
"model": "
|
| 3 |
"answers": {
|
| 4 |
"scene": {
|
| 5 |
"type": "choice",
|
|
|
|
| 1 |
{
|
| 2 |
+
"model": "jevvision",
|
| 3 |
"answers": {
|
| 4 |
"scene": {
|
| 5 |
"type": "choice",
|
runtime/kev/api.py
CHANGED
|
@@ -64,7 +64,7 @@ class SystemOneRequest(BaseModel):
|
|
| 64 |
model_config = ConfigDict(hide_input_in_errors=True)
|
| 65 |
|
| 66 |
state: JSONContent
|
| 67 |
-
model: str = "
|
| 68 |
questions: dict[str, Question] = Field(min_length=1)
|
| 69 |
images: list[ImageInput] | None = Field(default=None, min_length=1, max_length=MAX_IMAGES) # omitted or null = text-only
|
| 70 |
|
|
|
|
| 64 |
model_config = ConfigDict(hide_input_in_errors=True)
|
| 65 |
|
| 66 |
state: JSONContent
|
| 67 |
+
model: str = "jevvision"
|
| 68 |
questions: dict[str, Question] = Field(min_length=1)
|
| 69 |
images: list[ImageInput] | None = Field(default=None, min_length=1, max_length=MAX_IMAGES) # omitted or null = text-only
|
| 70 |
|
runtime/kev/jevvision.py
CHANGED
|
@@ -124,7 +124,7 @@ class JEVision:
|
|
| 124 |
state,
|
| 125 |
questions: dict,
|
| 126 |
images: list[dict[str, str]] | None = None,
|
| 127 |
-
model: str = "
|
| 128 |
) -> dict:
|
| 129 |
"""Return one KEV/System One response dict for the supplied request."""
|
| 130 |
request = SystemOneRequest(
|
|
|
|
| 124 |
state,
|
| 125 |
questions: dict,
|
| 126 |
images: list[dict[str, str]] | None = None,
|
| 127 |
+
model: str = "jevvision",
|
| 128 |
) -> dict:
|
| 129 |
"""Return one KEV/System One response dict for the supplied request."""
|
| 130 |
request = SystemOneRequest(
|
runtime/kev/serve.py
CHANGED
|
@@ -129,7 +129,7 @@ def prepare(req):
|
|
| 129 |
return req.model_copy(update={"state": with_date_facts(req.state)}) if DATE_FACTS else req
|
| 130 |
|
| 131 |
|
| 132 |
-
app = FastAPI(title="
|
| 133 |
app.add_middleware(CORSMiddleware, allow_origins=["*"], allow_methods=["*"], allow_headers=["*"])
|
| 134 |
|
| 135 |
|
|
@@ -194,8 +194,9 @@ def systemone_separate(req: SystemOneRequest):
|
|
| 194 |
@app.get("/v1/models")
|
| 195 |
def models():
|
| 196 |
s = server()
|
| 197 |
-
return {"models": [{"id": "
|
| 198 |
-
"lora": s.checkpoint.meta.lora, "
|
|
|
|
| 199 |
"context_tokens": s.context_limit,
|
| 200 |
"prefix_cache": {"size": PREFIX_CACHE_SIZE, "min_state_tokens": s.prefix_min_tokens, "hits": s.prefix_hits,
|
| 201 |
"max_state_tokens": PREFIX_CACHE_MAX_TOKENS, "misses": s.prefix_misses,
|
|
|
|
| 129 |
return req.model_copy(update={"state": with_date_facts(req.state)}) if DATE_FACTS else req
|
| 130 |
|
| 131 |
|
| 132 |
+
app = FastAPI(title="JEVision")
|
| 133 |
app.add_middleware(CORSMiddleware, allow_origins=["*"], allow_methods=["*"], allow_headers=["*"])
|
| 134 |
|
| 135 |
|
|
|
|
| 194 |
@app.get("/v1/models")
|
| 195 |
def models():
|
| 196 |
s = server()
|
| 197 |
+
return {"models": [{"id": "jevvision", "aliases": [], "run": s.checkpoint.requested, "base": s.checkpoint.meta.base,
|
| 198 |
+
"lora": s.checkpoint.meta.lora, "visual_route": "JEVision visual sidecar" if s.visual is not None else None,
|
| 199 |
+
"device": s.device, "backend": s.model.backend, "dtype": s.model.dtype, "temperature": s.model.head.temperature,
|
| 200 |
"context_tokens": s.context_limit,
|
| 201 |
"prefix_cache": {"size": PREFIX_CACHE_SIZE, "min_state_tokens": s.prefix_min_tokens, "hits": s.prefix_hits,
|
| 202 |
"max_state_tokens": PREFIX_CACHE_MAX_TOKENS, "misses": s.prefix_misses,
|