File size: 5,360 Bytes
7595189
e1c1c45
7595189
 
e1c1c45
7595189
e1c1c45
7595189
e1c1c45
 
 
 
 
f5a7ece
e1c1c45
7595189
 
e1c1c45
 
7595189
 
 
 
 
 
 
 
 
 
 
 
 
e1c1c45
 
7595189
 
 
 
 
 
f5a7ece
e1c1c45
 
f5a7ece
7595189
e1c1c45
7595189
e1c1c45
7595189
 
 
 
e1c1c45
 
 
 
f5a7ece
 
e1c1c45
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
f5a7ece
 
e1c1c45
f5a7ece
 
 
 
 
 
 
 
 
 
e1c1c45
 
f5a7ece
 
e1c1c45
f5a7ece
e1c1c45
 
f5a7ece
7595189
f5a7ece
e1c1c45
f5a7ece
7595189
e1c1c45
f5a7ece
e1c1c45
7595189
f5a7ece
 
7595189
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
import gradio as gr
import pandas as pd
from transformers import pipeline
import re
import os

# 1. โหลดโมเดล NER (เหมือนเดิม)
print("กำลังโหลดโมเดล...")
hf_token = os.getenv("HF_TOKEN")
ner_pipeline = pipeline(
    "token-classification", 
    model="loolootech/no-name-ner-th", 
    device=-1,
    token=hf_token
)
print("โมเดลพร้อมใช้งานแล้ว")


# 2. ฟังก์ชันสำหรับรวม Token (เหมือนเดิม)
def merge_entities(ner_results):
    merged_entities = []
    current_entity = None
    for entity in ner_results:
        entity_type = re.sub(r'^[BI]-', '', entity['entity'])
        if current_entity and entity['start'] == current_entity['end'] and entity_type == current_entity['type']:
            current_entity['word'] += entity['word']
            current_entity['end'] = entity['end']
            current_entity['score'] = max(current_entity['score'], entity['score'])
        else:
            if current_entity:
                merged_entities.append(current_entity)
            current_entity = {
                'type': entity_type, 'word': entity['word'],
                'start': entity['start'], 'end': entity['end'], 'score': entity['score']
            }
    if current_entity:
        merged_entities.append(current_entity)
    return merged_entities


# 3. ฟังก์ชันสำหรับ De-identification ของข้อความ 1 บรรทัด (เหมือนเดิม)
def deidentify_single_text(text):
    if pd.isna(text) or not isinstance(text, str) or not text.strip():
        return ""

    ner_results = ner_pipeline(text)
    merged = merge_entities(ner_results)
    
    redacted_text = text
    for entity in reversed(merged):
        start, end, label = entity['start'], entity['end'], entity['type']
        redacted_text = redacted_text[:start] + f"[{label}]" + redacted_text[end:]
        
    return redacted_text


# 4. [อัปเดต] ฟังก์ชันสำหรับประมวลผลไฟล์ (ไม่ต้องรับชื่อคอลัมน์แล้ว)
def process_entire_file(uploaded_file, progress=gr.Progress(track_tqdm=True)):
    if uploaded_file is None:
        raise gr.Error("กรุณาอัปโหลดไฟล์ก่อน")

    file_path = uploaded_file.name
    
    # อ่านไฟล์ด้วย Pandas
    try:
        if file_path.endswith('.csv'):
            df = pd.read_csv(file_path)
        elif file_path.endswith(('.xlsx', '.xls')):
            df = pd.read_excel(file_path)
        else:
            raise gr.Error("ไฟล์ไม่รองรับ กรุณาอัปโหลด .csv หรือ .xlsx เท่านั้น")
    except Exception as e:
        raise gr.Error(f"ไม่สามารถอ่านไฟล์ได้: {e}")

    # สร้าง DataFrame ใหม่สำหรับเก็บผลลัพธ์
    df_redacted = df.copy()

    # [Key Change] ค้นหาคอลัมน์ทั้งหมดที่มีข้อมูลเป็นประเภทข้อความ (object)
    text_columns = df.select_dtypes(include=['object']).columns

    if len(text_columns) == 0:
        raise gr.Error("ไม่พบคอลัมน์ที่เป็นข้อมูลประเภทข้อความ (text) ในไฟล์นี้เลย")

    # วนลูปและประมวลผลทุกคอลัมน์ที่หาเจอ
    print(f"กำลังประมวลผลคอลัมน์: {list(text_columns)}")
    for col_name in progress.tqdm(text_columns, desc="Processing text columns"):
        df_redacted[col_name] = df[col_name].astype(str).apply(deidentify_single_text)
    
    # สร้างไฟล์ผลลัพธ์เพื่อให้ผู้ใช้ดาวน์โหลด
    output_filepath = "processed_output_full.csv"
    df_redacted.to_csv(output_filepath, index=False, encoding='utf-8-sig')
    
    return df_redacted, output_filepath


# 5. [อัปเดต] สร้างหน้าเว็บ Gradio (ตัดช่องใส่ชื่อคอลัมน์ออก)
iface = gr.Interface(
    fn=process_entire_file,
    inputs=[
        gr.File(label="อัปโหลดไฟล์ CSV หรือ Excel ที่ต้องการตรวจสอบทั้งตาราง", file_types=[".csv", ".xlsx", ".xls"])
    ],
    outputs=[
        gr.DataFrame(label="ตารางผลลัพธ์ (Output Table Preview)", wrap=True, max_rows=10),
        gr.File(label="ดาวน์โหลดผลลัพธ์ (Download Result as CSV)")
    ],
    title="📁 Automatic Table De-identification",
    description="อัปโหลดไฟล์ตาราง (CSV, Excel) แล้วระบบจะค้นหาคอลัมน์ที่เป็น 'ข้อความ' ทั้งหมดโดยอัตโนมัติ และทำการปกปิดข้อมูลส่วนบุคคลให้ทันที",
    allow_flagging="never"
)

# รันแอป
iface.launch()