{ "cells": [ { "cell_type": "code", "execution_count": 1, "id": "03c454b7", "metadata": {}, "outputs": [], "source": [ "import pickle\n", "from sklearn.feature_extraction.text import TfidfVectorizer\n", "from sklearn.metrics.pairwise import cosine_similarity\n", "import pandas as pd\n", "import numpy as np" ] }, { "cell_type": "code", "execution_count": 2, "id": "a9fb1906", "metadata": {}, "outputs": [], "source": [ "events = pickle.load(open(\"../model/events.pkl\", 'rb'))\n", "users = pickle.load(open(\"../model/users.pkl\", 'rb'))" ] }, { "cell_type": "code", "execution_count": 3, "id": "7eb63efe", "metadata": {}, "outputs": [], "source": [ "users_event = pickle.load(open(\"../model/users_event.pkl\", 'rb'))" ] }, { "cell_type": "code", "execution_count": 5, "id": "8ed9e2a6", "metadata": {}, "outputs": [], "source": [ "event_similarity = pickle.load(open(\"../model/events_similarity.pkl\", 'rb'))\n", "users_similarity = pickle.load(open(\"../model/users_similarity.pkl\", 'rb'))" ] }, { "cell_type": "code", "execution_count": 6, "id": "62ac57ac", "metadata": {}, "outputs": [], "source": [ "events_vector = pickle.load(open(\"../model/events_vector.pkl\", 'rb'))\n", "users_vector = pickle.load(open(\"../model/users_vector.pkl\", 'rb'))" ] }, { "cell_type": "code", "execution_count": 7, "id": "d64af43a", "metadata": {}, "outputs": [ { "data": { "text/html": [ "
\n", "\n", "\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "
user_idnamekey
068e0e01c6fbdb8a90b76129bJacob JenningsGame Development Blockchain FinTech Cloud Comp...
168e0e01c6fbdb8a90b76129cMrs. Julie MartinezSports Entrepreneurship DevOps Robotics Cultur...
\n", "
" ], "text/plain": [ " user_id name \\\n", "0 68e0e01c6fbdb8a90b76129b Jacob Jennings \n", "1 68e0e01c6fbdb8a90b76129c Mrs. Julie Martinez \n", "\n", " key \n", "0 Game Development Blockchain FinTech Cloud Comp... \n", "1 Sports Entrepreneurship DevOps Robotics Cultur... " ] }, "execution_count": 7, "metadata": {}, "output_type": "execute_result" } ], "source": [ "users.head(2)" ] }, { "cell_type": "code", "execution_count": 8, "id": "e352269e", "metadata": {}, "outputs": [ { "data": { "text/html": [ "
\n", "\n", "\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "
event_idkey
068e0d68abee8a0e226e6fe96versatile homogeneous interface jaclynborough ...
168e0d68abee8a0e226e6fe97intuitive empowering orchestration new mariabe...
\n", "
" ], "text/plain": [ " event_id key\n", "0 68e0d68abee8a0e226e6fe96 versatile homogeneous interface jaclynborough ...\n", "1 68e0d68abee8a0e226e6fe97 intuitive empowering orchestration new mariabe..." ] }, "execution_count": 8, "metadata": {}, "output_type": "execute_result" } ], "source": [ "events.head(2)" ] }, { "cell_type": "code", "execution_count": 9, "id": "d06b051a", "metadata": {}, "outputs": [], "source": [ "def similar_user(user_id):\n", " index = users[users['user_id']== user_id].index[0]\n", " distances = users_similarity[index]\n", " users_list = sorted(list(enumerate(distances)),reverse=True , key=lambda x:x[1])[1:6]\n", "\n", " similar_users=[]\n", " for i in users_list:\n", " similar_users.append(users.iloc[i[0]].user_id)\n", "\n", " return similar_users" ] }, { "cell_type": "code", "execution_count": 10, "id": "a78114a8", "metadata": {}, "outputs": [], "source": [ "a=similar_user('68e0e01c6fbdb8a90b76129b')" ] }, { "cell_type": "code", "execution_count": 11, "id": "5cb62ec0", "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "['68e0e01c6fbdb8a90b761330', '68e0e01c6fbdb8a90b7613cf', '68e0e01c6fbdb8a90b761434', '68e0e01c6fbdb8a90b761319', '68e0e01c6fbdb8a90b7612cc']\n" ] } ], "source": [ "print(a)" ] }, { "cell_type": "code", "execution_count": 12, "id": "53890ce4", "metadata": {}, "outputs": [ { "data": { "text/html": [ "
\n", "\n", "\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "
user_idevents
068e0e01c6fbdb8a90b76129b[Game Development, Blockchain, FinTech, Cloud ...
168e0e01c6fbdb8a90b76129c[Sports, Entrepreneurship, DevOps, Robotics, C...
268e0e01c6fbdb8a90b76129d[Robotics, DevOps, EdTech, Leadership, React]
368e0e01c6fbdb8a90b76129e[AI/ML, FinTech, DevOps, EdTech, Game Developm...
\n", "
" ], "text/plain": [ " user_id events\n", "0 68e0e01c6fbdb8a90b76129b [Game Development, Blockchain, FinTech, Cloud ...\n", "1 68e0e01c6fbdb8a90b76129c [Sports, Entrepreneurship, DevOps, Robotics, C...\n", "2 68e0e01c6fbdb8a90b76129d [Robotics, DevOps, EdTech, Leadership, React]\n", "3 68e0e01c6fbdb8a90b76129e [AI/ML, FinTech, DevOps, EdTech, Game Developm..." ] }, "execution_count": 12, "metadata": {}, "output_type": "execute_result" } ], "source": [ "users_event.head(4)" ] }, { "cell_type": "code", "execution_count": 13, "id": "d4702cb5", "metadata": {}, "outputs": [], "source": [ "def events_of_user(user_id):\n", " for i in range(len(users_event)):\n", " if users_event['user_id'][i]==user_id:\n", " return users_event['events'][i]\n", " return None\n" ] }, { "cell_type": "code", "execution_count": 14, "id": "c2064203", "metadata": {}, "outputs": [], "source": [ "#Collaborative Filtering\n", "\n", "def recommend(user_id):\n", " similar_users = similar_user(user_id)\n", " my_events = events_of_user(user_id)\n", " other_events = []\n", " for i in similar_users:\n", " ev = events_of_user(i)\n", " for j in ev:\n", " if j not in my_events:\n", " other_events.append(j)\n", " return set(other_events)\n", " " ] }, { "cell_type": "code", "execution_count": 15, "id": "6b6abe49", "metadata": {}, "outputs": [ { "data": { "text/plain": [ "11" ] }, "execution_count": 15, "metadata": {}, "output_type": "execute_result" } ], "source": [ "len(recommend('68e0e01c6fbdb8a90b76129b'))" ] }, { "cell_type": "code", "execution_count": 16, "id": "445008c4", "metadata": {}, "outputs": [], "source": [ "# Content-Based Filtering Implementation\n", "\n", "def content_based_filtering(user_id, num_recommendations=5):\n", " \n", " # Get user index\n", " user_index = users[users['user_id'] == user_id].index[0]\n", " \n", " # Create TF-IDF vectors for events (fit on events data)\n", " event_tfidf = TfidfVectorizer(stop_words='english', max_features=5000)\n", " event_vectors = event_tfidf.fit_transform(events['key'])\n", " \n", " # Transform user profile using the same vectorizer\n", " user_profile = users.iloc[user_index]['key']\n", " user_vector = event_tfidf.transform([user_profile])\n", " \n", " # Calculate cosine similarity between user profile and all events\n", " similarity_scores = cosine_similarity(user_vector, event_vectors).flatten()\n", " \n", " # Get top recommendations\n", " top_indices = similarity_scores.argsort()[-num_recommendations:][::-1]\n", " \n", " # Create results dataframe\n", " recommended_events = events.iloc[top_indices].copy()\n", " recommended_events['similarity_score'] = similarity_scores[top_indices]\n", " \n", " return recommended_events[['event_id', 'similarity_score']]" ] }, { "cell_type": "code", "execution_count": 17, "id": "7552f8d3", "metadata": {}, "outputs": [], "source": [ "# Enhanced Content-Based Filtering with event filtering\n", "\n", "def content_based_filtering_enhanced(user_id, num_recommendations=5, exclude_attended=True):\n", " \"\"\"\n", " Enhanced content-based filtering that optionally excludes events the user has already attended\n", " \n", " Args:\n", " user_id: ID of the user to recommend events for\n", " num_recommendations: Number of events to recommend\n", " exclude_attended: Whether to exclude events the user has already attended\n", " \n", " Returns:\n", " DataFrame containing recommended events with similarity scores\n", " \"\"\"\n", " # Get user index\n", " user_index = users[users['user_id'] == user_id].index[0]\n", " \n", " # Get events the user has already attended (if excluding)\n", " attended_events = []\n", " if exclude_attended:\n", " attended_events = events_of_user(user_id) or []\n", " \n", " # Create TF-IDF vectors for events\n", " event_tfidf = TfidfVectorizer(stop_words='english', max_features=5000, ngram_range=(1, 2))\n", " event_vectors = event_tfidf.fit_transform(events['key'])\n", " \n", " # Transform user profile using the same vectorizer\n", " user_profile = users.iloc[user_index]['key']\n", " user_vector = event_tfidf.transform([user_profile])\n", " \n", " # Calculate cosine similarity between user profile and all events\n", " similarity_scores = cosine_similarity(user_vector, event_vectors).flatten()\n", " \n", " # Filter out attended events if requested\n", " if exclude_attended and attended_events:\n", " for event_id in attended_events:\n", " event_idx = events[events['event_id'] == event_id].index\n", " if len(event_idx) > 0:\n", " similarity_scores[event_idx[0]] = -1 # Set to -1 to exclude from top recommendations\n", " \n", " # Get top recommendations\n", " top_indices = similarity_scores.argsort()[-num_recommendations:][::-1]\n", " \n", " # Create results dataframe\n", " recommended_events = events.iloc[top_indices].copy()\n", " recommended_events['similarity_score'] = similarity_scores[top_indices]\n", " \n", " return recommended_events[['event_id', 'similarity_score']]" ] }, { "cell_type": "code", "execution_count": 18, "id": "bc343d4e", "metadata": {}, "outputs": [], "source": [ "# Hybrid Recommendation System (Collaborative + Content-Based)\n", "\n", "def hybrid_recommendation(user_id, num_recommendations=10, collaborative_weight=0.6, content_weight=0.4):\n", " \"\"\"\n", " Hybrid recommendation system combining collaborative filtering and content-based filtering\n", " \n", " Args:\n", " user_id: ID of the user to recommend events for\n", " num_recommendations: Number of events to recommend\n", " collaborative_weight: Weight for collaborative filtering (0-1)\n", " content_weight: Weight for content-based filtering (0-1)\n", " \n", " Returns:\n", " DataFrame containing recommended events with combined scores\n", " \"\"\"\n", " # Get collaborative filtering recommendations\n", " collab_events = recommend(user_id) # This returns a set of event_ids\n", " \n", " # Get content-based recommendations \n", " content_recs = content_based_filtering_enhanced(user_id, num_recommendations=20, exclude_attended=True)\n", " \n", " # Create a scoring system\n", " event_scores = {}\n", " \n", " # Score content-based recommendations\n", " for _, row in content_recs.iterrows():\n", " event_id = row['event_id']\n", " content_score = row['similarity_score']\n", " event_scores[event_id] = content_weight * content_score\n", " \n", " # Boost scores for collaborative filtering recommendations\n", " collab_boost = collaborative_weight / len(collab_events) if collab_events else 0\n", " for event_id in collab_events:\n", " if event_id in event_scores:\n", " event_scores[event_id] += collab_boost\n", " else:\n", " event_scores[event_id] = collab_boost\n", " \n", " # Sort by combined score and get top recommendations\n", " sorted_events = sorted(event_scores.items(), key=lambda x: x[1], reverse=True)[:num_recommendations]\n", " \n", " # Create results dataframe\n", " recommended_event_ids = [event_id for event_id, _ in sorted_events]\n", " recommended_scores = [score for _, score in sorted_events]\n", " \n", " # Get event details\n", " result_events = events[events['event_id'].isin(recommended_event_ids)].copy()\n", " \n", " # Add scores to the result\n", " score_dict = dict(sorted_events)\n", " result_events['hybrid_score'] = result_events['event_id'].map(score_dict)\n", " result_events = result_events.sort_values('hybrid_score', ascending=False)\n", " \n", " return result_events[['event_id']]" ] }, { "cell_type": "code", "execution_count": 19, "id": "80fb9772", "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ " event_id\n", "226 68e0d79a3048252621b7d89f\n", "358 68e0d79a3048252621b7d923\n", "38 68e0d7821f7dc4e22f39d51a\n", "301 68e0d79a3048252621b7d8ea\n", "27 68e0d7821f7dc4e22f39d50f\n", "60 68e0d7821f7dc4e22f39d530\n", "378 68e0d79a3048252621b7d937\n", "324 68e0d79a3048252621b7d901\n" ] } ], "source": [ "hybrid_recs = hybrid_recommendation('68e0e01c6fbdb8a90b76129b', num_recommendations=8)\n", "print(hybrid_recs)" ] } ], "metadata": { "kernelspec": { "display_name": "venv", "language": "python", "name": "python3" }, "language_info": { "codemirror_mode": { "name": "ipython", "version": 3 }, "file_extension": ".py", "mimetype": "text/x-python", "name": "python", "nbconvert_exporter": "python", "pygments_lexer": "ipython3", "version": "3.11.14" } }, "nbformat": 4, "nbformat_minor": 5 }