import numpy as np import pandas as pd from scipy.sparse import csr_matrix from sklearn.metrics.pairwise import cosine_similarity # Build a sparse user × article matrix (1 = clicked, 0 = didn't) matrix = csr_matrix((np.ones(len(clicks)), (user_rows, article_cols)), shape=(n_users, n_articles)) def recommend(user_id, matrix, top_n=15, n_neighbors=50): """Find 50 most similar users and rank the articles they clicked that our user hasn't seen yet.""" u = user_idx[user_id] # Cosine similarity between this user and everyone else sims = cosine_similarity(matrix[u], matrix).flatten() sims[u] = 0 # don't recommend to yourself # Take the top 50 most similar users top_neighbors = np.argsort(sims)[-n_neighbors:][::-1] weights = sims[top_neighbors] # Score articles by weighted sum of neighbour clicks scores = np.asarray(matrix[top_neighbors].T.dot(weights)).flatten() # Zero out articles the user already clicked scores[matrix[u].toarray().flatten() > 0] = 0 # Return the top-scoring articles top_articles = np.argsort(scores)[-top_n:][::-1] return top_articles