from rank_bm25 import BM25Okapi import re def tokenize(text): # Lowercase and split on non-alphanumeric characters return re.findall(r'\b[a-z]+\b', text.lower()) raw_documents = [ "Machine learning algorithms learn patterns from data", "Deep learning uses neural networks with many layers", "Natural language processing handles text and speech", "Computer vision processes images and video", "Neural networks are inspired by the human brain" ] tokenized_docs = [tokenize(doc) for doc in raw_documents] bm25 = BM25Okapi(tokenized_docs) query_tokens = tokenize("neural networks and deep learning") scores = bm25.get_scores(query_tokens) ranked = sorted(zip(scores, raw_documents), reverse=True) for score, doc in ranked: print(f"{score:.3f} — {doc}")