Instructions to use petros/bert-base-cypriot-uncased-v1 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use petros/bert-base-cypriot-uncased-v1 with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("fill-mask", model="petros/bert-base-cypriot-uncased-v1")# Load model directly from transformers import AutoTokenizer, AutoModelForMaskedLM tokenizer = AutoTokenizer.from_pretrained("petros/bert-base-cypriot-uncased-v1") model = AutoModelForMaskedLM.from_pretrained("petros/bert-base-cypriot-uncased-v1", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Update README.md
Browse files
README.md
CHANGED
|
@@ -42,10 +42,10 @@ def strip_accents_and_lowercase(s):
|
|
| 42 |
return ''.join(c for c in unicodedata.normalize('NFD', s)
|
| 43 |
if unicodedata.category(c) != 'Mn').lower()
|
| 44 |
|
| 45 |
-
accented_string = "
|
| 46 |
unaccented_string = strip_accents_and_lowercase(accented_string)
|
| 47 |
|
| 48 |
-
print(unaccented_string) #
|
| 49 |
|
| 50 |
```
|
| 51 |
|
|
@@ -71,35 +71,21 @@ tokenizer_cypriot = AutoTokenizer.from_pretrained('petros/bert-base-cypriot-unca
|
|
| 71 |
lm_model_cypriot = AutoModelWithLMHead.from_pretrained('petros/bert-base-cypriot-uncased-v1')
|
| 72 |
|
| 73 |
# ================ EXAMPLE 1 ================
|
| 74 |
-
text_1 = '
|
| 75 |
-
# EN: 'Sees the [MASK].'
|
| 76 |
input_ids = tokenizer_cypriot.encode(text_1)
|
| 77 |
-
print(tokenizer_cypriot.convert_ids_to_tokens(input_ids))
|
| 78 |
-
|
| 79 |
outputs = lm_model_cypriot(torch.tensor([input_ids]))[0]
|
| 80 |
-
print(
|
| 81 |
-
# the most plausible prediction for [MASK] is "song"
|
| 82 |
|
| 83 |
# ================ EXAMPLE 2 ================
|
| 84 |
-
text_2 = '
|
| 85 |
-
# EN: 'He is a [MASK] person.'
|
| 86 |
input_ids = tokenizer_greek.encode(text_2)
|
| 87 |
-
print(tokenizer_greek.convert_ids_to_tokens(input_ids))
|
| 88 |
-
|
| 89 |
-
outputs = lm_model_greek(torch.tensor([input_ids]))[0]
|
| 90 |
-
print(tokenizer_greek.convert_ids_to_tokens(outputs[0, 3].max(0)[1].item()))
|
| 91 |
-
# the most plausible prediction for [MASK] is "good"
|
| 92 |
-
|
| 93 |
-
# ================ EXAMPLE 3 ================
|
| 94 |
-
text_3 = 'Είναι ένας [MASK] άνθρωπος και κάνει συχνά [MASK].'
|
| 95 |
-
# EN: 'He is a [MASK] person he does frequently [MASK].'
|
| 96 |
-
input_ids = tokenizer_greek.encode(text_3)
|
| 97 |
-
print(tokenizer_greek.convert_ids_to_tokens(input_ids))
|
| 98 |
-
# ['[CLS]', 'ειναι', 'ενας', '[MASK]', 'ανθρωπος', 'και', 'κανει', 'συχνα', '[MASK]', '.', '[SEP]']
|
| 99 |
outputs = lm_model_greek(torch.tensor([input_ids]))[0]
|
| 100 |
-
print(tokenizer_greek.convert_ids_to_tokens(outputs[0,
|
| 101 |
-
|
| 102 |
-
```
|
| 103 |
|
| 104 |
About Me
|
| 105 |
Petros Andreou
|
|
|
|
| 42 |
return ''.join(c for c in unicodedata.normalize('NFD', s)
|
| 43 |
if unicodedata.category(c) != 'Mn').lower()
|
| 44 |
|
| 45 |
+
accented_string = "Τούτη εν η Κυπριακή έκδοση του BERT."
|
| 46 |
unaccented_string = strip_accents_and_lowercase(accented_string)
|
| 47 |
|
| 48 |
+
print(unaccented_string) # τουτη εν η κυπριακη εκδοση του bert.
|
| 49 |
|
| 50 |
```
|
| 51 |
|
|
|
|
| 71 |
lm_model_cypriot = AutoModelWithLMHead.from_pretrained('petros/bert-base-cypriot-uncased-v1')
|
| 72 |
|
| 73 |
# ================ EXAMPLE 1 ================
|
| 74 |
+
text_1 = 'Τι [MASK] ρε'
|
|
|
|
| 75 |
input_ids = tokenizer_cypriot.encode(text_1)
|
| 76 |
+
print(tokenizer_cypriot.convert_ids_to_tokens(input_ids)) # ['[CLS]', 'τι', '[MASK]', 'ρε', '[SEP]']
|
| 77 |
+
|
| 78 |
outputs = lm_model_cypriot(torch.tensor([input_ids]))[0]
|
| 79 |
+
print(tokenizer_cypriot.convert_ids_to_tokens(outputs[0, 2].max(0)[1].item())) # ειδους
|
|
|
|
| 80 |
|
| 81 |
# ================ EXAMPLE 2 ================
|
| 82 |
+
text_2 = 'Eίσαι μια [MASK].'
|
|
|
|
| 83 |
input_ids = tokenizer_greek.encode(text_2)
|
| 84 |
+
print(tokenizer_greek.convert_ids_to_tokens(input_ids)) #['[CLS]', 'eισ', '##αι', 'μια', '[MASK]', '.', '[SEP]']
|
| 85 |
+
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 86 |
outputs = lm_model_greek(torch.tensor([input_ids]))[0]
|
| 87 |
+
print(tokenizer_greek.convert_ids_to_tokens(outputs[0, 3].max(0)[1].item())) # χαρα
|
| 88 |
+
|
|
|
|
| 89 |
|
| 90 |
About Me
|
| 91 |
Petros Andreou
|