from rank_bm25 import BM25Okapi # Your documents, pre-tokenized (split into word lists) documents = [ "machine learning algorithms learn patterns from data".split(), "deep learning uses neural networks with many layers".split(), "natural language processing handles text and speech".split(), "computer vision processes images and video".split(), "neural networks are inspired by the human brain".split() ] # Build the BM25 index bm25 = BM25Okapi(documents) # Query query = "neural networks deep learning".split() # Get scores for each document scores = bm25.get_scores(query) print(scores) # Get the top N most relevant documents top_docs = bm25.get_top_n(query, documents, n=2) for doc in top_docs: print(" ".join(doc))