from gensim.models import Word2Vec # Small demo corpus sentences = [ ["the", "quick", "brown", "fox", "jumps"], ["the", "lazy", "dog", "sleeps"], ["the", "fox", "outsmarts", "the", "dog"], ] # Train a skip-gram Word2Vec model model = Word2Vec( sentences, # list of tokenized sentences vector_size=50, # embedding dimensions window=2, # context window size min_count=1, # include all words sg=1, # 1 for skip-gram; 0 for CBOW ) # Find most similar words to “fox” similar_words = model.wv.most_similar("fox", topn=3) for word, score in similar_words: print(word, score)