petros commited on
Commit
a584f86
·
1 Parent(s): 6b1b19a

Update README.md

Browse files
Files changed (1) hide show
  1. README.md +11 -25
README.md CHANGED
@@ -42,10 +42,10 @@ def strip_accents_and_lowercase(s):
42
  return ''.join(c for c in unicodedata.normalize('NFD', s)
43
  if unicodedata.category(c) != 'Mn').lower()
44
 
45
- accented_string = "Αυτή είναι η Ελληνική έκδοση του BERT."
46
  unaccented_string = strip_accents_and_lowercase(accented_string)
47
 
48
- print(unaccented_string) # αυτη ειναι η ελληνικη εκδοση του bert.
49
 
50
  ```
51
 
@@ -71,35 +71,21 @@ tokenizer_cypriot = AutoTokenizer.from_pretrained('petros/bert-base-cypriot-unca
71
  lm_model_cypriot = AutoModelWithLMHead.from_pretrained('petros/bert-base-cypriot-uncased-v1')
72
 
73
  # ================ EXAMPLE 1 ================
74
- text_1 = 'Θώρει τη [MASK].'
75
- # EN: 'Sees the [MASK].'
76
  input_ids = tokenizer_cypriot.encode(text_1)
77
- print(tokenizer_cypriot.convert_ids_to_tokens(input_ids))
78
- # ['[CLS]', 'o', 'ποιητης', 'εγραψε', 'ενα', '[MASK]', '.', '[SEP]']
79
  outputs = lm_model_cypriot(torch.tensor([input_ids]))[0]
80
- print(tokenizer_greek.convert_ids_to_tokens(outputs[0, 5].max(0)[1].item()))
81
- # the most plausible prediction for [MASK] is "song"
82
 
83
  # ================ EXAMPLE 2 ================
84
- text_2 = 'Είναι ένας [MASK] άνθρωπος.'
85
- # EN: 'He is a [MASK] person.'
86
  input_ids = tokenizer_greek.encode(text_2)
87
- print(tokenizer_greek.convert_ids_to_tokens(input_ids))
88
- # ['[CLS]', 'ειναι', 'ενας', '[MASK]', 'ανθρωπος', '.', '[SEP]']
89
- outputs = lm_model_greek(torch.tensor([input_ids]))[0]
90
- print(tokenizer_greek.convert_ids_to_tokens(outputs[0, 3].max(0)[1].item()))
91
- # the most plausible prediction for [MASK] is "good"
92
-
93
- # ================ EXAMPLE 3 ================
94
- text_3 = 'Είναι ένας [MASK] άνθρωπος και κάνει συχνά [MASK].'
95
- # EN: 'He is a [MASK] person he does frequently [MASK].'
96
- input_ids = tokenizer_greek.encode(text_3)
97
- print(tokenizer_greek.convert_ids_to_tokens(input_ids))
98
- # ['[CLS]', 'ειναι', 'ενας', '[MASK]', 'ανθρωπος', 'και', 'κανει', 'συχνα', '[MASK]', '.', '[SEP]']
99
  outputs = lm_model_greek(torch.tensor([input_ids]))[0]
100
- print(tokenizer_greek.convert_ids_to_tokens(outputs[0, 8].max(0)[1].item()))
101
- # the most plausible prediction for the second [MASK] is "trips"
102
- ```
103
 
104
  About Me
105
  Petros Andreou
 
42
  return ''.join(c for c in unicodedata.normalize('NFD', s)
43
  if unicodedata.category(c) != 'Mn').lower()
44
 
45
+ accented_string = "Τούτη εν η Κυπριακή έκδοση του BERT."
46
  unaccented_string = strip_accents_and_lowercase(accented_string)
47
 
48
+ print(unaccented_string) # τουτη εν η κυπριακη εκδοση του bert.
49
 
50
  ```
51
 
 
71
  lm_model_cypriot = AutoModelWithLMHead.from_pretrained('petros/bert-base-cypriot-uncased-v1')
72
 
73
  # ================ EXAMPLE 1 ================
74
+ text_1 = 'Τι [MASK] ρε'
 
75
  input_ids = tokenizer_cypriot.encode(text_1)
76
+ print(tokenizer_cypriot.convert_ids_to_tokens(input_ids)) # ['[CLS]', 'τι', '[MASK]', 'ρε', '[SEP]']
77
+
78
  outputs = lm_model_cypriot(torch.tensor([input_ids]))[0]
79
+ print(tokenizer_cypriot.convert_ids_to_tokens(outputs[0, 2].max(0)[1].item())) # ειδους
 
80
 
81
  # ================ EXAMPLE 2 ================
82
+ text_2 = 'Eίσαι μια [MASK].'
 
83
  input_ids = tokenizer_greek.encode(text_2)
84
+ print(tokenizer_greek.convert_ids_to_tokens(input_ids)) #['[CLS]', 'eισ', '##αι', 'μια', '[MASK]', '.', '[SEP]']
85
+
 
 
 
 
 
 
 
 
 
 
86
  outputs = lm_model_greek(torch.tensor([input_ids]))[0]
87
+ print(tokenizer_greek.convert_ids_to_tokens(outputs[0, 3].max(0)[1].item())) # χαρα
88
+
 
89
 
90
  About Me
91
  Petros Andreou