# WygLore Leaf — Korean NER endpoint (HF Space) # # Gradio 테스트 UI (/) + 깔끔한 FastAPI JSON API (/extract · /extract_batch · /health). # 플러그인은 /extract 를 host.nativeFetch 로 *한 방* 호출 (모바일 저메모리 폴백). # 무거운 모델 추론만 여기서; canon-앵커 / POS 필터 / assemble 은 플러그인 JS 가 그대로. # # ※ fastapi/uvicorn/pydantic 은 gradio 의존성으로 이미 설치됨 -> requirements 불요. import gradio as gr from fastapi import FastAPI, HTTPException from fastapi.middleware.cors import CORSMiddleware from pydantic import BaseModel from huggingface_hub import list_repo_files from optimum.onnxruntime import ORTModelForTokenClassification from transformers import AutoTokenizer, pipeline MODEL_REPO = "m6dd8m/wl-ko-ner-v2" # ONNX 가중치 (operator 공개) TOK_REPO = "monologg/koelectra-base-v3-discriminator" # 토크나이저 (파인튜닝해도 동일) # 1) repo 안의 .onnx 자동 탐지 — fp32 우선 (서버 CPU 라 fp16 shader 불요). onnx_files = [f for f in list_repo_files(MODEL_REPO) if f.endswith(".onnx")] if not onnx_files: raise RuntimeError(f"{MODEL_REPO} 에 .onnx 가 없음 — repo 내용 확인") onnx_path = next((f for f in onnx_files if "fp32" in f.lower()), onnx_files[0]) subfolder, onnx_name = (onnx_path.rsplit("/", 1) if "/" in onnx_path else ("", onnx_path)) print(f"[boot] onnx -> '{onnx_path}'") # 2) 모델 + 토크나이저 + pipeline. model = ORTModelForTokenClassification.from_pretrained( MODEL_REPO, file_name=onnx_name, subfolder=subfolder) tok = AutoTokenizer.from_pretrained(TOK_REPO) print(f"[boot] id2label = {model.config.id2label}") ner = pipeline("token-classification", model=model, tokenizer=tok, aggregation_strategy="simple") # 3) pipeline 출력 -> on-device ner.js 와 동일 모양 (type/text/score/start/end). def _spans(entities): return [ {"text": e["word"], "type": e["entity_group"], # PS / LC / OG / DT / TI / QT "score": float(e["score"]), "start": int(e["start"]), "end": int(e["end"])} for e in entities ] def extract(text): if not text or not text.strip(): return [] return _spans(ner(text)) # ───────────────────────── FastAPI (깔끔한 JSON API) ───────────────────────── app = FastAPI(title="WygLore Leaf Korean NER") app.add_middleware(CORSMiddleware, allow_origins=["*"], allow_methods=["*"], allow_headers=["*"]) class OneReq(BaseModel): text: str class BatchReq(BaseModel): texts: list[str] @app.get("/health") def health(): return {"ok": True, "model": MODEL_REPO, "onnx": onnx_path} @app.post("/extract") # 한 방: {"text": "..."} -> {"entities": [...]} def extract_api(req: OneReq): return {"entities": extract(req.text)} @app.post("/extract_batch") # 배치(콜드스타트 효율): {"texts": [...]} -> {"results": [[...], ...]} def extract_batch_api(req: BatchReq): if len(req.texts) > 512: # 가벼운 abuse/메모리 가드 raise HTTPException(400, "batch too large (max 512)") idx = [i for i, t in enumerate(req.texts) if t and t.strip()] outs = ner([req.texts[i] for i in idx]) if idx else [] results = [[] for _ in req.texts] for k, i in enumerate(idx): results[i] = _spans(outs[k]) return {"results": results} # ───────────────────────── Gradio 테스트 UI 를 / 에 마운트 ───────────────────────── demo = gr.Interface( fn=extract, inputs=gr.Textbox(lines=4, label="한국어 텍스트", placeholder="새봄이 여의도에서 안도현을 만났다."), outputs=gr.JSON(label="엔티티 스팬"), title="WygLore Leaf — Korean NER (wl-ko-ner-v2)", description="UI=테스트용 · API=POST /extract · 배치=/extract_batch · 문서=/docs", ) app = gr.mount_gradio_app(app, demo, path="/") # API 라우트가 먼저 등록돼 우선 매칭 if __name__ == "__main__": import uvicorn uvicorn.run(app, host="0.0.0.0", port=7860) # HF Space 기대 포트