import string import unicodedata # We can use "_" to represent an out-of-vocabulary character, that is, any character we are not handling in our model allowed_characters = string.ascii_letters + " .,;'" + "_" n_letters = len(allowed_characters) # Turn a Unicode string to plain ASCII, thanks to https://stackoverflow.com/a/518232/2809427 def unicodeToAscii(s): return ''.join( c for c in unicodedata.normalize('NFD', s) if unicodedata.category(c) != 'Mn' and c in allowed_characters )