NanoVDR-Demo / app.py
Ryenhails's picture
List NanoVDR first; NanoVDR is the default model
1184bf3 verified
Raw History Blame Contribute Delete
7.74 kB
"""NanoVDR / ColNanoVDR interactive retrieval demo.
Both students are text-only query encoders that run on CPU against a page index
their vision-language teacher built offline; no vision model runs at query time.
- NanoVDR (single-vector): 69M DistilBERT student of Qwen3-VL-Embedding-2B,
one vector scored by cosine similarity against the teacher's page vectors.
- ColNanoVDR (multi-vector): 149M Ettin student of ColQwen3.5-4.5B, weighted
token vectors scored by MaxSim against the teacher's page tokens.
Corpus: ViDoRe v3 Computer Science (1,360 pages).
"""
import json
import os
import time
import gradio as gr
import numpy as np
import torch
from sentence_transformers import MultiVectorEncoder, SentenceTransformer
DATA = "data"
torch.set_num_threads(max(1, os.cpu_count() or 1))
with open(os.path.join(DATA, "queries.json")) as f:
EXAMPLES = json.load(f)
BY_TEXT = {q["text"]: q for q in EXAMPLES}
# ── ColNanoVDR: multi-vector index by ColQwen3.5-4.5B ─────────────────────────
mv_docs = torch.from_numpy(np.load(os.path.join(DATA, "colnanovdr", "doc_emb.npy")).astype(np.float32))
mv_len = torch.from_numpy(np.load(os.path.join(DATA, "colnanovdr", "doc_lengths.npy")))
mv_mask = torch.arange(mv_docs.shape[1])[None, :] < mv_len[:, None]
mv_model = MultiVectorEncoder("nanovdr/ColNanoVDR-Q-Ettin150M-ColQwen35-320-ML", trust_remote_code=True, device="cpu")
# ── NanoVDR: single-vector index by Qwen3-VL-Embedding-2B ─────────────────────
sv_docs = np.load(os.path.join(DATA, "corpus_embeddings.npy")).astype(np.float32)
sv_docs /= np.maximum(np.linalg.norm(sv_docs, axis=1, keepdims=True), 1e-12)
sv_model = SentenceTransformer("nanovdr/NanoVDR-Q-DistilBERT-Qwen3VL2B-2048-ML", device="cpu")
N_PAGES = sv_docs.shape[0]
NANOVDR, COLNANOVDR = "NanoVDR (single-vector, 69M)", "ColNanoVDR (multi-vector, 149M)"
ABOUT = {
COLNANOVDR: (
"**ColNanoVDR** ([model](https://huggingface.co/nanovdr/ColNanoVDR-Q-Ettin150M-ColQwen35-320-ML), "
"[paper](https://arxiv.org/abs/2609.34899)): a 149M text-only encoder turns the query into one vector per token, "
"each scaled by a learned weight, and **MaxSim** scores them against page tokens that "
"[ColQwen3.5-4.5B](https://huggingface.co/athrael-soju/colqwen3.5-4.5B-v3) encoded offline."
),
NANOVDR: (
"**NanoVDR** ([model](https://huggingface.co/nanovdr/NanoVDR-Q-DistilBERT-Qwen3VL2B-2048-ML), "
"[paper](https://arxiv.org/abs/2603.12824)): a 69M text-only encoder turns the query into a single vector, "
"scored by **cosine similarity** against page vectors that "
"[Qwen3-VL-Embedding-2B](https://huggingface.co/Qwen/Qwen3-VL-Embedding-2B) encoded offline."
),
}
TEACHER = {COLNANOVDR: ("colnanovdr", "ColQwen3.5-4.5B"), NANOVDR: ("nanovdr", "Qwen3-VL-Embedding-2B")}
for m in (mv_model, sv_model):
m.encode_query(["warm up"]) if m is mv_model else m.encode(["warm up"])
print(f"Ready: {N_PAGES} pages; multi-vector index {tuple(mv_docs.shape)}, single-vector index {sv_docs.shape}")
def page(i):
return os.path.join(DATA, "images", f"{i:04d}.jpg")
def token_weights(query, vecs):
"""Pair each non-special token with its learned weight (the norm of its vector)."""
special = set(mv_model.tokenizer.all_special_ids)
toks = [mv_model.tokenizer.decode([t]) for t in mv_model.tokenizer(query)["input_ids"] if t not in special]
w = vecs.norm(dim=-1).tolist()
if len(toks) != len(w):
return [(query, None)]
top = max(w) or 1.0
return [(t, round(x / top, 2)) for t, x in zip(toks, w)]
def search(query, top_k, model_name):
query, top_k = (query or "").strip(), int(top_k)
if not query:
return [], "Please enter a query.", [], [], ""
t0 = time.perf_counter()
if model_name == COLNANOVDR:
q = mv_model.encode_query([query], convert_to_numpy=False)[0].float()
t_enc = (time.perf_counter() - t0) * 1000
t0 = time.perf_counter()
with torch.no_grad():
sim = torch.einsum("nld,kd->nlk", mv_docs, q).masked_fill(~mv_mask[:, :, None], float("-inf"))
scores = sim.amax(1).sum(-1).numpy()
weights, what = token_weights(query, q), f"{q.shape[0]} weighted token vectors Β· MaxSim"
else:
q = sv_model.encode([query], normalize_embeddings=True)[0]
t_enc = (time.perf_counter() - t0) * 1000
t0 = time.perf_counter()
scores = sv_docs @ q
weights, what = [], "1 vector Β· cosine similarity"
ranked = np.argsort(-scores)[:top_k].tolist()
t_score = (time.perf_counter() - t0) * 1000
ex = BY_TEXT.get(query)
rel = set(ex["relevant"]) if ex else set()
cap = lambda r, i: f"#{r + 1} page {i}" + (" βœ“ relevant" if i in rel else "")
student = [(page(i), cap(r, i)) for r, i in enumerate(ranked)]
info = f"Query encoded in **{t_enc:.0f} ms** on CPU ({what} over {N_PAGES:,} pages in {t_score:.0f} ms)"
teacher, note = [], ""
if ex:
key, tname = TEACHER[model_name]
tt = ex["teacher_top10"][key][:top_k]
teacher = [(page(i), cap(r, i)) for r, i in enumerate(tt)]
note = (f"Example query from ViDoRe v3: the {tname} teacher's own ranking is shown below for comparison "
f"(precomputed). Overlap with the student's top-{top_k}: **{len(set(tt) & set(ranked))}/{top_k}**. "
f"Pages marked βœ“ are annotated as relevant.")
return student, info, weights, teacher, note
def switch(model_name):
return ABOUT[model_name], gr.update(visible=model_name == COLNANOVDR)
with gr.Blocks(title="NanoVDR Demo") as demo:
gr.Markdown(
f"""
# NanoVDR: visual document retrieval with text-only query encoders
Type a query. A small **text-only** student encodes it on **CPU** and is scored against a page index that its
vision-language teacher built offline; the index is used unchanged and no vision model runs at query time.
Choose the single-vector **NanoVDR** or the multi-vector **ColNanoVDR** student below.
**Corpus**: {N_PAGES:,} pages from [ViDoRe v3 Computer Science](https://huggingface.co/datasets/vidore/vidore_v3_computer_science)
(academic papers, slides, diagrams).
"""
)
model_in = gr.Radio([NANOVDR, COLNANOVDR], value=NANOVDR, label="Model")
about = gr.Markdown(ABOUT[NANOVDR])
with gr.Row():
query_in = gr.Textbox(label="Query", lines=2, scale=4,
placeholder="e.g., How does the V-model integrate testing compared with Waterfall?")
top_k = gr.Slider(1, 10, value=5, step=1, label="Top-K", scale=1)
btn = gr.Button("Search", variant="primary")
info = gr.Markdown()
weights = gr.HighlightedText(label="Learned token weights (relative to the largest)", combine_adjacent=False,
show_legend=False, visible=False)
gallery = gr.Gallery(label="Student: retrieved pages", columns=5, height="auto", object_fit="contain")
note = gr.Markdown()
teacher_gallery = gr.Gallery(label="Teacher (example queries only, precomputed)", columns=5, height="auto",
object_fit="contain")
gr.Examples(examples=[[q["text"]] for q in EXAMPLES[:12]], inputs=[query_in], label="Example queries (ViDoRe v3)")
outs = [gallery, info, weights, teacher_gallery, note]
btn.click(search, [query_in, top_k, model_in], outs)
query_in.submit(search, [query_in, top_k, model_in], outs)
model_in.change(switch, [model_in], [about, weights]).then(search, [query_in, top_k, model_in], outs)
demo.launch(theme=gr.themes.Soft())