chethan999's picture
Upload app.py with huggingface_hub
16d0283 verified
Raw
History Blame Contribute Delete
2.29 kB
import gradio as gr
import sentencepiece as spm
MODEL_PREFIX = "household_power_bpe"
# Load SentencePiece model
sp = spm.SentencePieceProcessor(model_file=f"{MODEL_PREFIX}.model")
def tokenize_text(text):
"""Tokenize input text and return token information"""
if not text:
return "Please enter some text", ""
# Encode
token_ids = sp.encode(text, out_type=int)
token_strings = sp.encode(text, out_type=str)
# Decode
decoded = sp.decode(token_ids)
# Format output
token_info = f"**Number of tokens:** {len(token_ids)}\n\n"
token_info += f"**Token IDs:**\n{token_ids}\n\n"
token_info += f"**Token Strings:**\n{token_strings}"
return token_info, decoded
# Example texts
examples = [
["DATE=16/12/2006|TIME=17:24:00|GAP=4.216|GRP=0.418|V=234.840|GI=18.400|SM1=0.000|SM2=1.000|SM3=17.000"],
["DATE=01/01/2007|TIME=00:00:00|GAP=5.123|GRP=1.234|V=240.000|GI=20.000|SM1=1.000|SM2=2.000|SM3=15.000"],
["GAP=3.500|GRP=0.250|V=230.000"]
]
# Create Gradio interface
with gr.Blocks(title="Household Power BPE Tokenizer") as demo:
gr.Markdown("# Household Power BPE Tokenizer")
gr.Markdown("This tokenizer is trained on household power consumption data using BPE (Byte-Pair Encoding)")
gr.Markdown("**Vocabulary Size:** 8000 tokens | **Model Type:** BPE")
with gr.Row():
with gr.Column():
input_text = gr.Textbox(
label="Input Text",
placeholder="Enter text to tokenize (e.g., DATE=16/12/2006|TIME=17:24:00|GAP=4.216...)",
lines=5
)
tokenize_btn = gr.Button("Tokenize", variant="primary")
with gr.Column():
token_output = gr.Markdown(label="Token Information")
decoded_output = gr.Textbox(
label="Decoded Text",
lines=3
)
gr.Examples(
examples=examples,
inputs=input_text,
label="Example Inputs"
)
tokenize_btn.click(
fn=tokenize_text,
inputs=input_text,
outputs=[token_output, decoded_output]
)
input_text.submit(
fn=tokenize_text,
inputs=input_text,
outputs=[token_output, decoded_output]
)
if __name__ == "__main__":
demo.launch()