asilvamaia commited on
Commit
68464cb
·
verified ·
1 Parent(s): f0d8cd9

Update src/streamlit_app.py

Browse files
Files changed (1) hide show
  1. src/streamlit_app.py +22 -7
src/streamlit_app.py CHANGED
@@ -5,17 +5,24 @@ from urllib.parse import urlparse
5
  import io
6
 
7
  # --- CONFIGURAÇÕES ---
8
- # ⚠️ MANTENHA O ID DO SEU MODELO AQUI
 
9
  MODEL_ID = "asilvamaia/ident_br"
10
 
11
  st.set_page_config(page_title="Validador .BR", page_icon="🇧🇷")
12
 
13
- # --- FUNÇÃO DE LIMPEZA (V11 - IGUAL AO LOCAL) ---
14
  def limpar_entrada(texto: str) -> str:
 
 
 
15
  texto = str(texto).strip().lower()
16
  if not texto: return ""
 
 
17
  if "@" in texto: return texto
18
 
 
19
  if "http" not in texto and "://" not in texto:
20
  texto_temp = "http://" + texto
21
  else:
@@ -25,9 +32,11 @@ def limpar_entrada(texto: str) -> str:
25
  parsed = urlparse(texto_temp)
26
  dominio_limpo = parsed.netloc if parsed.netloc else texto
27
 
 
28
  if ":" in dominio_limpo:
29
  dominio_limpo = dominio_limpo.split(':')[0]
30
 
 
31
  if dominio_limpo.startswith("www."):
32
  dominio_limpo = dominio_limpo[4:]
33
 
@@ -35,11 +44,13 @@ def limpar_entrada(texto: str) -> str:
35
  except:
36
  return texto
37
 
 
38
  @st.cache_resource
39
  def load_model():
40
  try:
41
  tokenizer = AutoTokenizer.from_pretrained(MODEL_ID)
42
  model = AutoModelForSequenceClassification.from_pretrained(MODEL_ID)
 
43
  model.to("cpu")
44
  model.eval()
45
  return tokenizer, model
@@ -49,11 +60,14 @@ def load_model():
49
 
50
  tokenizer, model = load_model()
51
 
 
52
  st.title("🇧🇷 Validador de Domínios .BR")
 
53
 
54
- uploaded_file = st.file_uploader("Carregar lista suja (.txt)", type="txt")
55
 
56
  if uploaded_file and tokenizer:
 
57
  stringio = io.StringIO(uploaded_file.getvalue().decode("utf-8"))
58
  linhas = stringio.readlines()
59
 
@@ -61,12 +75,13 @@ if uploaded_file and tokenizer:
61
  validos = []
62
  rejeitados = []
63
 
 
64
  progress_bar = st.progress(0)
65
  total_linhas = len(linhas)
66
 
67
  for i, linha in enumerate(linhas):
68
  original = linha.strip()
69
- # Pula linhas de metadados do arquivo (ex: )
70
- if not original or original.startswith("`, ``.
71
- * O script local pode estar classificando isso como "Lixo" (Rejeitado).
72
- * Adicionei uma linha no código acima (`if original.startswith("
 
5
  import io
6
 
7
  # --- CONFIGURAÇÕES ---
8
+ # ⚠️ IMPORTANTE: Substitua pelo ID do seu modelo no Hugging Face
9
+ # Exemplo: "joao/classificador-dominios-br"
10
  MODEL_ID = "asilvamaia/ident_br"
11
 
12
  st.set_page_config(page_title="Validador .BR", page_icon="🇧🇷")
13
 
14
+ # --- FUNÇÃO DE LIMPEZA (V11 - Igual ao script local) ---
15
  def limpar_entrada(texto: str) -> str:
16
+ """
17
+ Remove sujeira, portas, protocolos e www.
18
+ """
19
  texto = str(texto).strip().lower()
20
  if not texto: return ""
21
+
22
+ # Mantém e-mail intacto para o modelo rejeitar explicitamente
23
  if "@" in texto: return texto
24
 
25
+ # Garante protocolo para o urlparse funcionar corretamente
26
  if "http" not in texto and "://" not in texto:
27
  texto_temp = "http://" + texto
28
  else:
 
32
  parsed = urlparse(texto_temp)
33
  dominio_limpo = parsed.netloc if parsed.netloc else texto
34
 
35
+ # Remove a porta (:8080)
36
  if ":" in dominio_limpo:
37
  dominio_limpo = dominio_limpo.split(':')[0]
38
 
39
+ # Remove o 'www.' do início
40
  if dominio_limpo.startswith("www."):
41
  dominio_limpo = dominio_limpo[4:]
42
 
 
44
  except:
45
  return texto
46
 
47
+ # --- CARREGAMENTO DO MODELO (Com Cache) ---
48
  @st.cache_resource
49
  def load_model():
50
  try:
51
  tokenizer = AutoTokenizer.from_pretrained(MODEL_ID)
52
  model = AutoModelForSequenceClassification.from_pretrained(MODEL_ID)
53
+ # Força uso de CPU no Space gratuito para evitar erros de memória/driver
54
  model.to("cpu")
55
  model.eval()
56
  return tokenizer, model
 
60
 
61
  tokenizer, model = load_model()
62
 
63
+ # --- INTERFACE DO USUÁRIO ---
64
  st.title("🇧🇷 Validador de Domínios .BR")
65
+ st.write("Faça upload de uma lista suja (.txt) para extrair apenas domínios .br válidos.")
66
 
67
+ uploaded_file = st.file_uploader("Carregar arquivo .txt", type="txt")
68
 
69
  if uploaded_file and tokenizer:
70
+ # Lê o arquivo enviado
71
  stringio = io.StringIO(uploaded_file.getvalue().decode("utf-8"))
72
  linhas = stringio.readlines()
73
 
 
75
  validos = []
76
  rejeitados = []
77
 
78
+ # Barra de progresso
79
  progress_bar = st.progress(0)
80
  total_linhas = len(linhas)
81
 
82
  for i, linha in enumerate(linhas):
83
  original = linha.strip()
84
+
85
+ # --- CORREÇÃO DO ERRO ANTERIOR AQUI ---
86
+ # Ignora linhas vazias ou metadados do arquivo como
87
+ if not original or original.startswith(")