Spaces:
Running
Running
Download app.py from nanovdr/NanoVDR-Demo: direct link, hf CLI and curl.
- Browser
- Download file 7.74 kB
-
https://huggingface.co/spaces/nanovdr/NanoVDR-Demo/resolve/main/app.py
- Command line
-
hf download hf://spaces/nanovdr/NanoVDR-Demo/app.py
-
curl -L -o app.py https://huggingface.co/spaces/nanovdr/NanoVDR-Demo/resolve/main/app.py
7.74 kB
| """NanoVDR / ColNanoVDR interactive retrieval demo. | |
| Both students are text-only query encoders that run on CPU against a page index | |
| their vision-language teacher built offline; no vision model runs at query time. | |
| - NanoVDR (single-vector): 69M DistilBERT student of Qwen3-VL-Embedding-2B, | |
| one vector scored by cosine similarity against the teacher's page vectors. | |
| - ColNanoVDR (multi-vector): 149M Ettin student of ColQwen3.5-4.5B, weighted | |
| token vectors scored by MaxSim against the teacher's page tokens. | |
| Corpus: ViDoRe v3 Computer Science (1,360 pages). | |
| """ | |
| import json | |
| import os | |
| import time | |
| import gradio as gr | |
| import numpy as np | |
| import torch | |
| from sentence_transformers import MultiVectorEncoder, SentenceTransformer | |
| DATA = "data" | |
| torch.set_num_threads(max(1, os.cpu_count() or 1)) | |
| with open(os.path.join(DATA, "queries.json")) as f: | |
| EXAMPLES = json.load(f) | |
| BY_TEXT = {q["text"]: q for q in EXAMPLES} | |
| # ββ ColNanoVDR: multi-vector index by ColQwen3.5-4.5B βββββββββββββββββββββββββ | |
| mv_docs = torch.from_numpy(np.load(os.path.join(DATA, "colnanovdr", "doc_emb.npy")).astype(np.float32)) | |
| mv_len = torch.from_numpy(np.load(os.path.join(DATA, "colnanovdr", "doc_lengths.npy"))) | |
| mv_mask = torch.arange(mv_docs.shape[1])[None, :] < mv_len[:, None] | |
| mv_model = MultiVectorEncoder("nanovdr/ColNanoVDR-Q-Ettin150M-ColQwen35-320-ML", trust_remote_code=True, device="cpu") | |
| # ββ NanoVDR: single-vector index by Qwen3-VL-Embedding-2B βββββββββββββββββββββ | |
| sv_docs = np.load(os.path.join(DATA, "corpus_embeddings.npy")).astype(np.float32) | |
| sv_docs /= np.maximum(np.linalg.norm(sv_docs, axis=1, keepdims=True), 1e-12) | |
| sv_model = SentenceTransformer("nanovdr/NanoVDR-Q-DistilBERT-Qwen3VL2B-2048-ML", device="cpu") | |
| N_PAGES = sv_docs.shape[0] | |
| NANOVDR, COLNANOVDR = "NanoVDR (single-vector, 69M)", "ColNanoVDR (multi-vector, 149M)" | |
| ABOUT = { | |
| COLNANOVDR: ( | |
| "**ColNanoVDR** ([model](https://huggingface.co/nanovdr/ColNanoVDR-Q-Ettin150M-ColQwen35-320-ML), " | |
| "[paper](https://arxiv.org/abs/2609.34899)): a 149M text-only encoder turns the query into one vector per token, " | |
| "each scaled by a learned weight, and **MaxSim** scores them against page tokens that " | |
| "[ColQwen3.5-4.5B](https://huggingface.co/athrael-soju/colqwen3.5-4.5B-v3) encoded offline." | |
| ), | |
| NANOVDR: ( | |
| "**NanoVDR** ([model](https://huggingface.co/nanovdr/NanoVDR-Q-DistilBERT-Qwen3VL2B-2048-ML), " | |
| "[paper](https://arxiv.org/abs/2603.12824)): a 69M text-only encoder turns the query into a single vector, " | |
| "scored by **cosine similarity** against page vectors that " | |
| "[Qwen3-VL-Embedding-2B](https://huggingface.co/Qwen/Qwen3-VL-Embedding-2B) encoded offline." | |
| ), | |
| } | |
| TEACHER = {COLNANOVDR: ("colnanovdr", "ColQwen3.5-4.5B"), NANOVDR: ("nanovdr", "Qwen3-VL-Embedding-2B")} | |
| for m in (mv_model, sv_model): | |
| m.encode_query(["warm up"]) if m is mv_model else m.encode(["warm up"]) | |
| print(f"Ready: {N_PAGES} pages; multi-vector index {tuple(mv_docs.shape)}, single-vector index {sv_docs.shape}") | |
| def page(i): | |
| return os.path.join(DATA, "images", f"{i:04d}.jpg") | |
| def token_weights(query, vecs): | |
| """Pair each non-special token with its learned weight (the norm of its vector).""" | |
| special = set(mv_model.tokenizer.all_special_ids) | |
| toks = [mv_model.tokenizer.decode([t]) for t in mv_model.tokenizer(query)["input_ids"] if t not in special] | |
| w = vecs.norm(dim=-1).tolist() | |
| if len(toks) != len(w): | |
| return [(query, None)] | |
| top = max(w) or 1.0 | |
| return [(t, round(x / top, 2)) for t, x in zip(toks, w)] | |
| def search(query, top_k, model_name): | |
| query, top_k = (query or "").strip(), int(top_k) | |
| if not query: | |
| return [], "Please enter a query.", [], [], "" | |
| t0 = time.perf_counter() | |
| if model_name == COLNANOVDR: | |
| q = mv_model.encode_query([query], convert_to_numpy=False)[0].float() | |
| t_enc = (time.perf_counter() - t0) * 1000 | |
| t0 = time.perf_counter() | |
| with torch.no_grad(): | |
| sim = torch.einsum("nld,kd->nlk", mv_docs, q).masked_fill(~mv_mask[:, :, None], float("-inf")) | |
| scores = sim.amax(1).sum(-1).numpy() | |
| weights, what = token_weights(query, q), f"{q.shape[0]} weighted token vectors Β· MaxSim" | |
| else: | |
| q = sv_model.encode([query], normalize_embeddings=True)[0] | |
| t_enc = (time.perf_counter() - t0) * 1000 | |
| t0 = time.perf_counter() | |
| scores = sv_docs @ q | |
| weights, what = [], "1 vector Β· cosine similarity" | |
| ranked = np.argsort(-scores)[:top_k].tolist() | |
| t_score = (time.perf_counter() - t0) * 1000 | |
| ex = BY_TEXT.get(query) | |
| rel = set(ex["relevant"]) if ex else set() | |
| cap = lambda r, i: f"#{r + 1} page {i}" + (" β relevant" if i in rel else "") | |
| student = [(page(i), cap(r, i)) for r, i in enumerate(ranked)] | |
| info = f"Query encoded in **{t_enc:.0f} ms** on CPU ({what} over {N_PAGES:,} pages in {t_score:.0f} ms)" | |
| teacher, note = [], "" | |
| if ex: | |
| key, tname = TEACHER[model_name] | |
| tt = ex["teacher_top10"][key][:top_k] | |
| teacher = [(page(i), cap(r, i)) for r, i in enumerate(tt)] | |
| note = (f"Example query from ViDoRe v3: the {tname} teacher's own ranking is shown below for comparison " | |
| f"(precomputed). Overlap with the student's top-{top_k}: **{len(set(tt) & set(ranked))}/{top_k}**. " | |
| f"Pages marked β are annotated as relevant.") | |
| return student, info, weights, teacher, note | |
| def switch(model_name): | |
| return ABOUT[model_name], gr.update(visible=model_name == COLNANOVDR) | |
| with gr.Blocks(title="NanoVDR Demo") as demo: | |
| gr.Markdown( | |
| f""" | |
| # NanoVDR: visual document retrieval with text-only query encoders | |
| Type a query. A small **text-only** student encodes it on **CPU** and is scored against a page index that its | |
| vision-language teacher built offline; the index is used unchanged and no vision model runs at query time. | |
| Choose the single-vector **NanoVDR** or the multi-vector **ColNanoVDR** student below. | |
| **Corpus**: {N_PAGES:,} pages from [ViDoRe v3 Computer Science](https://huggingface.co/datasets/vidore/vidore_v3_computer_science) | |
| (academic papers, slides, diagrams). | |
| """ | |
| ) | |
| model_in = gr.Radio([NANOVDR, COLNANOVDR], value=NANOVDR, label="Model") | |
| about = gr.Markdown(ABOUT[NANOVDR]) | |
| with gr.Row(): | |
| query_in = gr.Textbox(label="Query", lines=2, scale=4, | |
| placeholder="e.g., How does the V-model integrate testing compared with Waterfall?") | |
| top_k = gr.Slider(1, 10, value=5, step=1, label="Top-K", scale=1) | |
| btn = gr.Button("Search", variant="primary") | |
| info = gr.Markdown() | |
| weights = gr.HighlightedText(label="Learned token weights (relative to the largest)", combine_adjacent=False, | |
| show_legend=False, visible=False) | |
| gallery = gr.Gallery(label="Student: retrieved pages", columns=5, height="auto", object_fit="contain") | |
| note = gr.Markdown() | |
| teacher_gallery = gr.Gallery(label="Teacher (example queries only, precomputed)", columns=5, height="auto", | |
| object_fit="contain") | |
| gr.Examples(examples=[[q["text"]] for q in EXAMPLES[:12]], inputs=[query_in], label="Example queries (ViDoRe v3)") | |
| outs = [gallery, info, weights, teacher_gallery, note] | |
| btn.click(search, [query_in, top_k, model_in], outs) | |
| query_in.submit(search, [query_in, top_k, model_in], outs) | |
| model_in.change(switch, [model_in], [about, weights]).then(search, [query_in, top_k, model_in], outs) | |
| demo.launch(theme=gr.themes.Soft()) | |