"""NanoVDR / ColNanoVDR interactive retrieval demo. Both students are text-only query encoders that run on CPU against a page index their vision-language teacher built offline; no vision model runs at query time. - NanoVDR (single-vector): 69M DistilBERT student of Qwen3-VL-Embedding-2B, one vector scored by cosine similarity against the teacher's page vectors. - ColNanoVDR (multi-vector): 149M Ettin student of ColQwen3.5-4.5B, weighted token vectors scored by MaxSim against the teacher's page tokens. Corpus: ViDoRe v3 Computer Science (1,360 pages). """ import json import os import time import gradio as gr import numpy as np import torch from sentence_transformers import MultiVectorEncoder, SentenceTransformer DATA = "data" torch.set_num_threads(max(1, os.cpu_count() or 1)) with open(os.path.join(DATA, "queries.json")) as f: EXAMPLES = json.load(f) BY_TEXT = {q["text"]: q for q in EXAMPLES} # ── ColNanoVDR: multi-vector index by ColQwen3.5-4.5B ───────────────────────── mv_docs = torch.from_numpy(np.load(os.path.join(DATA, "colnanovdr", "doc_emb.npy")).astype(np.float32)) mv_len = torch.from_numpy(np.load(os.path.join(DATA, "colnanovdr", "doc_lengths.npy"))) mv_mask = torch.arange(mv_docs.shape[1])[None, :] < mv_len[:, None] mv_model = MultiVectorEncoder("nanovdr/ColNanoVDR-Q-Ettin150M-ColQwen35-320-ML", trust_remote_code=True, device="cpu") # ── NanoVDR: single-vector index by Qwen3-VL-Embedding-2B ───────────────────── sv_docs = np.load(os.path.join(DATA, "corpus_embeddings.npy")).astype(np.float32) sv_docs /= np.maximum(np.linalg.norm(sv_docs, axis=1, keepdims=True), 1e-12) sv_model = SentenceTransformer("nanovdr/NanoVDR-Q-DistilBERT-Qwen3VL2B-2048-ML", device="cpu") N_PAGES = sv_docs.shape[0] NANOVDR, COLNANOVDR = "NanoVDR (single-vector, 69M)", "ColNanoVDR (multi-vector, 149M)" ABOUT = { COLNANOVDR: ( "**ColNanoVDR** ([model](https://huggingface.co/nanovdr/ColNanoVDR-Q-Ettin150M-ColQwen35-320-ML), " "[paper](https://arxiv.org/abs/2609.34899)): a 149M text-only encoder turns the query into one vector per token, " "each scaled by a learned weight, and **MaxSim** scores them against page tokens that " "[ColQwen3.5-4.5B](https://huggingface.co/athrael-soju/colqwen3.5-4.5B-v3) encoded offline." ), NANOVDR: ( "**NanoVDR** ([model](https://huggingface.co/nanovdr/NanoVDR-Q-DistilBERT-Qwen3VL2B-2048-ML), " "[paper](https://arxiv.org/abs/2603.12824)): a 69M text-only encoder turns the query into a single vector, " "scored by **cosine similarity** against page vectors that " "[Qwen3-VL-Embedding-2B](https://huggingface.co/Qwen/Qwen3-VL-Embedding-2B) encoded offline." ), } TEACHER = {COLNANOVDR: ("colnanovdr", "ColQwen3.5-4.5B"), NANOVDR: ("nanovdr", "Qwen3-VL-Embedding-2B")} for m in (mv_model, sv_model): m.encode_query(["warm up"]) if m is mv_model else m.encode(["warm up"]) print(f"Ready: {N_PAGES} pages; multi-vector index {tuple(mv_docs.shape)}, single-vector index {sv_docs.shape}") def page(i): return os.path.join(DATA, "images", f"{i:04d}.jpg") def token_weights(query, vecs): """Pair each non-special token with its learned weight (the norm of its vector).""" special = set(mv_model.tokenizer.all_special_ids) toks = [mv_model.tokenizer.decode([t]) for t in mv_model.tokenizer(query)["input_ids"] if t not in special] w = vecs.norm(dim=-1).tolist() if len(toks) != len(w): return [(query, None)] top = max(w) or 1.0 return [(t, round(x / top, 2)) for t, x in zip(toks, w)] def search(query, top_k, model_name): query, top_k = (query or "").strip(), int(top_k) if not query: return [], "Please enter a query.", [], [], "" t0 = time.perf_counter() if model_name == COLNANOVDR: q = mv_model.encode_query([query], convert_to_numpy=False)[0].float() t_enc = (time.perf_counter() - t0) * 1000 t0 = time.perf_counter() with torch.no_grad(): sim = torch.einsum("nld,kd->nlk", mv_docs, q).masked_fill(~mv_mask[:, :, None], float("-inf")) scores = sim.amax(1).sum(-1).numpy() weights, what = token_weights(query, q), f"{q.shape[0]} weighted token vectors · MaxSim" else: q = sv_model.encode([query], normalize_embeddings=True)[0] t_enc = (time.perf_counter() - t0) * 1000 t0 = time.perf_counter() scores = sv_docs @ q weights, what = [], "1 vector · cosine similarity" ranked = np.argsort(-scores)[:top_k].tolist() t_score = (time.perf_counter() - t0) * 1000 ex = BY_TEXT.get(query) rel = set(ex["relevant"]) if ex else set() cap = lambda r, i: f"#{r + 1} page {i}" + (" ✓ relevant" if i in rel else "") student = [(page(i), cap(r, i)) for r, i in enumerate(ranked)] info = f"Query encoded in **{t_enc:.0f} ms** on CPU ({what} over {N_PAGES:,} pages in {t_score:.0f} ms)" teacher, note = [], "" if ex: key, tname = TEACHER[model_name] tt = ex["teacher_top10"][key][:top_k] teacher = [(page(i), cap(r, i)) for r, i in enumerate(tt)] note = (f"Example query from ViDoRe v3: the {tname} teacher's own ranking is shown below for comparison " f"(precomputed). Overlap with the student's top-{top_k}: **{len(set(tt) & set(ranked))}/{top_k}**. " f"Pages marked ✓ are annotated as relevant.") return student, info, weights, teacher, note def switch(model_name): return ABOUT[model_name], gr.update(visible=model_name == COLNANOVDR) with gr.Blocks(title="NanoVDR Demo") as demo: gr.Markdown( f""" # NanoVDR: visual document retrieval with text-only query encoders Type a query. A small **text-only** student encodes it on **CPU** and is scored against a page index that its vision-language teacher built offline; the index is used unchanged and no vision model runs at query time. Choose the single-vector **NanoVDR** or the multi-vector **ColNanoVDR** student below. **Corpus**: {N_PAGES:,} pages from [ViDoRe v3 Computer Science](https://huggingface.co/datasets/vidore/vidore_v3_computer_science) (academic papers, slides, diagrams). """ ) model_in = gr.Radio([NANOVDR, COLNANOVDR], value=NANOVDR, label="Model") about = gr.Markdown(ABOUT[NANOVDR]) with gr.Row(): query_in = gr.Textbox(label="Query", lines=2, scale=4, placeholder="e.g., How does the V-model integrate testing compared with Waterfall?") top_k = gr.Slider(1, 10, value=5, step=1, label="Top-K", scale=1) btn = gr.Button("Search", variant="primary") info = gr.Markdown() weights = gr.HighlightedText(label="Learned token weights (relative to the largest)", combine_adjacent=False, show_legend=False, visible=False) gallery = gr.Gallery(label="Student: retrieved pages", columns=5, height="auto", object_fit="contain") note = gr.Markdown() teacher_gallery = gr.Gallery(label="Teacher (example queries only, precomputed)", columns=5, height="auto", object_fit="contain") gr.Examples(examples=[[q["text"]] for q in EXAMPLES[:12]], inputs=[query_in], label="Example queries (ViDoRe v3)") outs = [gallery, info, weights, teacher_gallery, note] btn.click(search, [query_in, top_k, model_in], outs) query_in.submit(search, [query_in, top_k, model_in], outs) model_in.change(switch, [model_in], [about, weights]).then(search, [query_in, top_k, model_in], outs) demo.launch(theme=gr.themes.Soft())