Download app.py from m6dd8m/wl-ko-ner: direct link, hf CLI and curl.
- Browser
- Download file 4.31 kB
-
https://huggingface.co/spaces/m6dd8m/wl-ko-ner/resolve/e833ff7552018fec4597c5f5b2b4334f0b1724c7/app.py
- Command line
-
hf download hf://spaces/m6dd8m/wl-ko-ner@e833ff7552018fec4597c5f5b2b4334f0b1724c7/app.py
-
curl -L -o app.py https://huggingface.co/spaces/m6dd8m/wl-ko-ner/resolve/e833ff7552018fec4597c5f5b2b4334f0b1724c7/app.py
4.31 kB
| # WygLore Leaf β Korean NER endpoint (HF Space) | |
| # | |
| # Gradio ν μ€νΈ UI (/) + κΉλν FastAPI JSON API (/extract Β· /extract_batch Β· /health). | |
| # νλ¬κ·ΈμΈμ /extract λ₯Ό host.nativeFetch λ‘ *ν λ°©* νΈμΆ (λͺ¨λ°μΌ μ λ©λͺ¨λ¦¬ ν΄λ°±). | |
| # λ¬΄κ±°μ΄ λͺ¨λΈ μΆλ‘ λ§ μ¬κΈ°μ; canon-μ΅μ»€ / POS νν° / assemble μ νλ¬κ·ΈμΈ JS κ° κ·Έλλ‘. | |
| # | |
| # β» fastapi/uvicorn/pydantic μ gradio μμ‘΄μ±μΌλ‘ μ΄λ―Έ μ€μΉλ¨ -> requirements λΆμ. | |
| import gradio as gr | |
| from fastapi import FastAPI, HTTPException | |
| from fastapi.middleware.cors import CORSMiddleware | |
| from pydantic import BaseModel | |
| from huggingface_hub import list_repo_files | |
| from optimum.onnxruntime import ORTModelForTokenClassification | |
| from transformers import AutoTokenizer, pipeline | |
| MODEL_REPO = "m6dd8m/wl-ko-ner-v2" # ONNX κ°μ€μΉ (operator 곡κ°) | |
| TOK_REPO = "monologg/koelectra-base-v3-discriminator" # ν ν¬λμ΄μ (νμΈνλν΄λ λμΌ) | |
| # 1) repo μμ .onnx μλ νμ§ β fp32 μ°μ (μλ² CPU λΌ fp16 shader λΆμ). | |
| onnx_files = [f for f in list_repo_files(MODEL_REPO) if f.endswith(".onnx")] | |
| if not onnx_files: | |
| raise RuntimeError(f"{MODEL_REPO} μ .onnx κ° μμ β repo λ΄μ© νμΈ") | |
| onnx_path = next((f for f in onnx_files if "fp32" in f.lower()), onnx_files[0]) | |
| subfolder, onnx_name = (onnx_path.rsplit("/", 1) if "/" in onnx_path else ("", onnx_path)) | |
| print(f"[boot] onnx -> '{onnx_path}'") | |
| # 2) λͺ¨λΈ + ν ν¬λμ΄μ + pipeline. | |
| model = ORTModelForTokenClassification.from_pretrained( | |
| MODEL_REPO, file_name=onnx_name, subfolder=subfolder) | |
| tok = AutoTokenizer.from_pretrained(TOK_REPO) | |
| print(f"[boot] id2label = {model.config.id2label}") | |
| ner = pipeline("token-classification", model=model, tokenizer=tok, | |
| aggregation_strategy="simple") | |
| # 3) pipeline μΆλ ₯ -> on-device ner.js μ λμΌ λͺ¨μ (type/text/score/start/end). | |
| def _spans(entities): | |
| return [ | |
| {"text": e["word"], | |
| "type": e["entity_group"], # PS / LC / OG / DT / TI / QT | |
| "score": float(e["score"]), | |
| "start": int(e["start"]), | |
| "end": int(e["end"])} | |
| for e in entities | |
| ] | |
| def extract(text): | |
| if not text or not text.strip(): | |
| return [] | |
| return _spans(ner(text)) | |
| # βββββββββββββββββββββββββ FastAPI (κΉλν JSON API) βββββββββββββββββββββββββ | |
| app = FastAPI(title="WygLore Leaf Korean NER") | |
| app.add_middleware(CORSMiddleware, allow_origins=["*"], | |
| allow_methods=["*"], allow_headers=["*"]) | |
| class OneReq(BaseModel): | |
| text: str | |
| class BatchReq(BaseModel): | |
| texts: list[str] | |
| def health(): | |
| return {"ok": True, "model": MODEL_REPO, "onnx": onnx_path} | |
| # ν λ°©: {"text": "..."} -> {"entities": [...]} | |
| def extract_api(req: OneReq): | |
| return {"entities": extract(req.text)} | |
| # λ°°μΉ(μ½λμ€ννΈ ν¨μ¨): {"texts": [...]} -> {"results": [[...], ...]} | |
| def extract_batch_api(req: BatchReq): | |
| if len(req.texts) > 512: # κ°λ²Όμ΄ abuse/λ©λͺ¨λ¦¬ κ°λ | |
| raise HTTPException(400, "batch too large (max 512)") | |
| idx = [i for i, t in enumerate(req.texts) if t and t.strip()] | |
| outs = ner([req.texts[i] for i in idx]) if idx else [] | |
| results = [[] for _ in req.texts] | |
| for k, i in enumerate(idx): | |
| results[i] = _spans(outs[k]) | |
| return {"results": results} | |
| # βββββββββββββββββββββββββ Gradio ν μ€νΈ UI λ₯Ό / μ λ§μ΄νΈ βββββββββββββββββββββββββ | |
| demo = gr.Interface( | |
| fn=extract, | |
| inputs=gr.Textbox(lines=4, label="νκ΅μ΄ ν μ€νΈ", | |
| placeholder="μλ΄μ΄ μ¬μλμμ μλνμ λ§λ¬λ€."), | |
| outputs=gr.JSON(label="μν°ν° μ€ν¬"), | |
| title="WygLore Leaf β Korean NER (wl-ko-ner-v2)", | |
| description="UI=ν μ€νΈμ© Β· API=POST /extract Β· λ°°μΉ=/extract_batch Β· λ¬Έμ=/docs", | |
| ) | |
| app = gr.mount_gradio_app(app, demo, path="/") # API λΌμ°νΈκ° λ¨Όμ λ±λ‘λΌ μ°μ λ§€μΉ | |
| if __name__ == "__main__": | |
| import uvicorn | |
| uvicorn.run(app, host="0.0.0.0", port=7860) # HF Space κΈ°λ ν¬νΈ | |