Spaces:
Runtime error
Runtime error
File size: 5,360 Bytes
7595189 e1c1c45 7595189 e1c1c45 7595189 e1c1c45 7595189 e1c1c45 f5a7ece e1c1c45 7595189 e1c1c45 7595189 e1c1c45 7595189 f5a7ece e1c1c45 f5a7ece 7595189 e1c1c45 7595189 e1c1c45 7595189 e1c1c45 f5a7ece e1c1c45 f5a7ece e1c1c45 f5a7ece e1c1c45 f5a7ece e1c1c45 f5a7ece e1c1c45 f5a7ece 7595189 f5a7ece e1c1c45 f5a7ece 7595189 e1c1c45 f5a7ece e1c1c45 7595189 f5a7ece 7595189 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 | import gradio as gr
import pandas as pd
from transformers import pipeline
import re
import os
# 1. โหลดโมเดล NER (เหมือนเดิม)
print("กำลังโหลดโมเดล...")
hf_token = os.getenv("HF_TOKEN")
ner_pipeline = pipeline(
"token-classification",
model="loolootech/no-name-ner-th",
device=-1,
token=hf_token
)
print("โมเดลพร้อมใช้งานแล้ว")
# 2. ฟังก์ชันสำหรับรวม Token (เหมือนเดิม)
def merge_entities(ner_results):
merged_entities = []
current_entity = None
for entity in ner_results:
entity_type = re.sub(r'^[BI]-', '', entity['entity'])
if current_entity and entity['start'] == current_entity['end'] and entity_type == current_entity['type']:
current_entity['word'] += entity['word']
current_entity['end'] = entity['end']
current_entity['score'] = max(current_entity['score'], entity['score'])
else:
if current_entity:
merged_entities.append(current_entity)
current_entity = {
'type': entity_type, 'word': entity['word'],
'start': entity['start'], 'end': entity['end'], 'score': entity['score']
}
if current_entity:
merged_entities.append(current_entity)
return merged_entities
# 3. ฟังก์ชันสำหรับ De-identification ของข้อความ 1 บรรทัด (เหมือนเดิม)
def deidentify_single_text(text):
if pd.isna(text) or not isinstance(text, str) or not text.strip():
return ""
ner_results = ner_pipeline(text)
merged = merge_entities(ner_results)
redacted_text = text
for entity in reversed(merged):
start, end, label = entity['start'], entity['end'], entity['type']
redacted_text = redacted_text[:start] + f"[{label}]" + redacted_text[end:]
return redacted_text
# 4. [อัปเดต] ฟังก์ชันสำหรับประมวลผลไฟล์ (ไม่ต้องรับชื่อคอลัมน์แล้ว)
def process_entire_file(uploaded_file, progress=gr.Progress(track_tqdm=True)):
if uploaded_file is None:
raise gr.Error("กรุณาอัปโหลดไฟล์ก่อน")
file_path = uploaded_file.name
# อ่านไฟล์ด้วย Pandas
try:
if file_path.endswith('.csv'):
df = pd.read_csv(file_path)
elif file_path.endswith(('.xlsx', '.xls')):
df = pd.read_excel(file_path)
else:
raise gr.Error("ไฟล์ไม่รองรับ กรุณาอัปโหลด .csv หรือ .xlsx เท่านั้น")
except Exception as e:
raise gr.Error(f"ไม่สามารถอ่านไฟล์ได้: {e}")
# สร้าง DataFrame ใหม่สำหรับเก็บผลลัพธ์
df_redacted = df.copy()
# [Key Change] ค้นหาคอลัมน์ทั้งหมดที่มีข้อมูลเป็นประเภทข้อความ (object)
text_columns = df.select_dtypes(include=['object']).columns
if len(text_columns) == 0:
raise gr.Error("ไม่พบคอลัมน์ที่เป็นข้อมูลประเภทข้อความ (text) ในไฟล์นี้เลย")
# วนลูปและประมวลผลทุกคอลัมน์ที่หาเจอ
print(f"กำลังประมวลผลคอลัมน์: {list(text_columns)}")
for col_name in progress.tqdm(text_columns, desc="Processing text columns"):
df_redacted[col_name] = df[col_name].astype(str).apply(deidentify_single_text)
# สร้างไฟล์ผลลัพธ์เพื่อให้ผู้ใช้ดาวน์โหลด
output_filepath = "processed_output_full.csv"
df_redacted.to_csv(output_filepath, index=False, encoding='utf-8-sig')
return df_redacted, output_filepath
# 5. [อัปเดต] สร้างหน้าเว็บ Gradio (ตัดช่องใส่ชื่อคอลัมน์ออก)
iface = gr.Interface(
fn=process_entire_file,
inputs=[
gr.File(label="อัปโหลดไฟล์ CSV หรือ Excel ที่ต้องการตรวจสอบทั้งตาราง", file_types=[".csv", ".xlsx", ".xls"])
],
outputs=[
gr.DataFrame(label="ตารางผลลัพธ์ (Output Table Preview)", wrap=True, max_rows=10),
gr.File(label="ดาวน์โหลดผลลัพธ์ (Download Result as CSV)")
],
title="📁 Automatic Table De-identification",
description="อัปโหลดไฟล์ตาราง (CSV, Excel) แล้วระบบจะค้นหาคอลัมน์ที่เป็น 'ข้อความ' ทั้งหมดโดยอัตโนมัติ และทำการปกปิดข้อมูลส่วนบุคคลให้ทันที",
allow_flagging="never"
)
# รันแอป
iface.launch() |