{
"cells": [
{
"cell_type": "code",
"execution_count": 1,
"id": "03c454b7",
"metadata": {},
"outputs": [],
"source": [
"import pickle\n",
"from sklearn.feature_extraction.text import TfidfVectorizer\n",
"from sklearn.metrics.pairwise import cosine_similarity\n",
"import pandas as pd\n",
"import numpy as np"
]
},
{
"cell_type": "code",
"execution_count": 2,
"id": "a9fb1906",
"metadata": {},
"outputs": [],
"source": [
"events = pickle.load(open(\"../model/events.pkl\", 'rb'))\n",
"users = pickle.load(open(\"../model/users.pkl\", 'rb'))"
]
},
{
"cell_type": "code",
"execution_count": 3,
"id": "7eb63efe",
"metadata": {},
"outputs": [],
"source": [
"users_event = pickle.load(open(\"../model/users_event.pkl\", 'rb'))"
]
},
{
"cell_type": "code",
"execution_count": 5,
"id": "8ed9e2a6",
"metadata": {},
"outputs": [],
"source": [
"event_similarity = pickle.load(open(\"../model/events_similarity.pkl\", 'rb'))\n",
"users_similarity = pickle.load(open(\"../model/users_similarity.pkl\", 'rb'))"
]
},
{
"cell_type": "code",
"execution_count": 6,
"id": "62ac57ac",
"metadata": {},
"outputs": [],
"source": [
"events_vector = pickle.load(open(\"../model/events_vector.pkl\", 'rb'))\n",
"users_vector = pickle.load(open(\"../model/users_vector.pkl\", 'rb'))"
]
},
{
"cell_type": "code",
"execution_count": 7,
"id": "d64af43a",
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"
\n",
"\n",
"
\n",
" \n",
" \n",
" | \n",
" user_id | \n",
" name | \n",
" key | \n",
"
\n",
" \n",
" \n",
" \n",
" | 0 | \n",
" 68e0e01c6fbdb8a90b76129b | \n",
" Jacob Jennings | \n",
" Game Development Blockchain FinTech Cloud Comp... | \n",
"
\n",
" \n",
" | 1 | \n",
" 68e0e01c6fbdb8a90b76129c | \n",
" Mrs. Julie Martinez | \n",
" Sports Entrepreneurship DevOps Robotics Cultur... | \n",
"
\n",
" \n",
"
\n",
"
"
],
"text/plain": [
" user_id name \\\n",
"0 68e0e01c6fbdb8a90b76129b Jacob Jennings \n",
"1 68e0e01c6fbdb8a90b76129c Mrs. Julie Martinez \n",
"\n",
" key \n",
"0 Game Development Blockchain FinTech Cloud Comp... \n",
"1 Sports Entrepreneurship DevOps Robotics Cultur... "
]
},
"execution_count": 7,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"users.head(2)"
]
},
{
"cell_type": "code",
"execution_count": 8,
"id": "e352269e",
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"\n",
"\n",
"
\n",
" \n",
" \n",
" | \n",
" event_id | \n",
" key | \n",
"
\n",
" \n",
" \n",
" \n",
" | 0 | \n",
" 68e0d68abee8a0e226e6fe96 | \n",
" versatile homogeneous interface jaclynborough ... | \n",
"
\n",
" \n",
" | 1 | \n",
" 68e0d68abee8a0e226e6fe97 | \n",
" intuitive empowering orchestration new mariabe... | \n",
"
\n",
" \n",
"
\n",
"
"
],
"text/plain": [
" event_id key\n",
"0 68e0d68abee8a0e226e6fe96 versatile homogeneous interface jaclynborough ...\n",
"1 68e0d68abee8a0e226e6fe97 intuitive empowering orchestration new mariabe..."
]
},
"execution_count": 8,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"events.head(2)"
]
},
{
"cell_type": "code",
"execution_count": 9,
"id": "d06b051a",
"metadata": {},
"outputs": [],
"source": [
"def similar_user(user_id):\n",
" index = users[users['user_id']== user_id].index[0]\n",
" distances = users_similarity[index]\n",
" users_list = sorted(list(enumerate(distances)),reverse=True , key=lambda x:x[1])[1:6]\n",
"\n",
" similar_users=[]\n",
" for i in users_list:\n",
" similar_users.append(users.iloc[i[0]].user_id)\n",
"\n",
" return similar_users"
]
},
{
"cell_type": "code",
"execution_count": 10,
"id": "a78114a8",
"metadata": {},
"outputs": [],
"source": [
"a=similar_user('68e0e01c6fbdb8a90b76129b')"
]
},
{
"cell_type": "code",
"execution_count": 11,
"id": "5cb62ec0",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"['68e0e01c6fbdb8a90b761330', '68e0e01c6fbdb8a90b7613cf', '68e0e01c6fbdb8a90b761434', '68e0e01c6fbdb8a90b761319', '68e0e01c6fbdb8a90b7612cc']\n"
]
}
],
"source": [
"print(a)"
]
},
{
"cell_type": "code",
"execution_count": 12,
"id": "53890ce4",
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"\n",
"\n",
"
\n",
" \n",
" \n",
" | \n",
" user_id | \n",
" events | \n",
"
\n",
" \n",
" \n",
" \n",
" | 0 | \n",
" 68e0e01c6fbdb8a90b76129b | \n",
" [Game Development, Blockchain, FinTech, Cloud ... | \n",
"
\n",
" \n",
" | 1 | \n",
" 68e0e01c6fbdb8a90b76129c | \n",
" [Sports, Entrepreneurship, DevOps, Robotics, C... | \n",
"
\n",
" \n",
" | 2 | \n",
" 68e0e01c6fbdb8a90b76129d | \n",
" [Robotics, DevOps, EdTech, Leadership, React] | \n",
"
\n",
" \n",
" | 3 | \n",
" 68e0e01c6fbdb8a90b76129e | \n",
" [AI/ML, FinTech, DevOps, EdTech, Game Developm... | \n",
"
\n",
" \n",
"
\n",
"
"
],
"text/plain": [
" user_id events\n",
"0 68e0e01c6fbdb8a90b76129b [Game Development, Blockchain, FinTech, Cloud ...\n",
"1 68e0e01c6fbdb8a90b76129c [Sports, Entrepreneurship, DevOps, Robotics, C...\n",
"2 68e0e01c6fbdb8a90b76129d [Robotics, DevOps, EdTech, Leadership, React]\n",
"3 68e0e01c6fbdb8a90b76129e [AI/ML, FinTech, DevOps, EdTech, Game Developm..."
]
},
"execution_count": 12,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"users_event.head(4)"
]
},
{
"cell_type": "code",
"execution_count": 13,
"id": "d4702cb5",
"metadata": {},
"outputs": [],
"source": [
"def events_of_user(user_id):\n",
" for i in range(len(users_event)):\n",
" if users_event['user_id'][i]==user_id:\n",
" return users_event['events'][i]\n",
" return None\n"
]
},
{
"cell_type": "code",
"execution_count": 14,
"id": "c2064203",
"metadata": {},
"outputs": [],
"source": [
"#Collaborative Filtering\n",
"\n",
"def recommend(user_id):\n",
" similar_users = similar_user(user_id)\n",
" my_events = events_of_user(user_id)\n",
" other_events = []\n",
" for i in similar_users:\n",
" ev = events_of_user(i)\n",
" for j in ev:\n",
" if j not in my_events:\n",
" other_events.append(j)\n",
" return set(other_events)\n",
" "
]
},
{
"cell_type": "code",
"execution_count": 15,
"id": "6b6abe49",
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"11"
]
},
"execution_count": 15,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"len(recommend('68e0e01c6fbdb8a90b76129b'))"
]
},
{
"cell_type": "code",
"execution_count": 16,
"id": "445008c4",
"metadata": {},
"outputs": [],
"source": [
"# Content-Based Filtering Implementation\n",
"\n",
"def content_based_filtering(user_id, num_recommendations=5):\n",
" \n",
" # Get user index\n",
" user_index = users[users['user_id'] == user_id].index[0]\n",
" \n",
" # Create TF-IDF vectors for events (fit on events data)\n",
" event_tfidf = TfidfVectorizer(stop_words='english', max_features=5000)\n",
" event_vectors = event_tfidf.fit_transform(events['key'])\n",
" \n",
" # Transform user profile using the same vectorizer\n",
" user_profile = users.iloc[user_index]['key']\n",
" user_vector = event_tfidf.transform([user_profile])\n",
" \n",
" # Calculate cosine similarity between user profile and all events\n",
" similarity_scores = cosine_similarity(user_vector, event_vectors).flatten()\n",
" \n",
" # Get top recommendations\n",
" top_indices = similarity_scores.argsort()[-num_recommendations:][::-1]\n",
" \n",
" # Create results dataframe\n",
" recommended_events = events.iloc[top_indices].copy()\n",
" recommended_events['similarity_score'] = similarity_scores[top_indices]\n",
" \n",
" return recommended_events[['event_id', 'similarity_score']]"
]
},
{
"cell_type": "code",
"execution_count": 17,
"id": "7552f8d3",
"metadata": {},
"outputs": [],
"source": [
"# Enhanced Content-Based Filtering with event filtering\n",
"\n",
"def content_based_filtering_enhanced(user_id, num_recommendations=5, exclude_attended=True):\n",
" \"\"\"\n",
" Enhanced content-based filtering that optionally excludes events the user has already attended\n",
" \n",
" Args:\n",
" user_id: ID of the user to recommend events for\n",
" num_recommendations: Number of events to recommend\n",
" exclude_attended: Whether to exclude events the user has already attended\n",
" \n",
" Returns:\n",
" DataFrame containing recommended events with similarity scores\n",
" \"\"\"\n",
" # Get user index\n",
" user_index = users[users['user_id'] == user_id].index[0]\n",
" \n",
" # Get events the user has already attended (if excluding)\n",
" attended_events = []\n",
" if exclude_attended:\n",
" attended_events = events_of_user(user_id) or []\n",
" \n",
" # Create TF-IDF vectors for events\n",
" event_tfidf = TfidfVectorizer(stop_words='english', max_features=5000, ngram_range=(1, 2))\n",
" event_vectors = event_tfidf.fit_transform(events['key'])\n",
" \n",
" # Transform user profile using the same vectorizer\n",
" user_profile = users.iloc[user_index]['key']\n",
" user_vector = event_tfidf.transform([user_profile])\n",
" \n",
" # Calculate cosine similarity between user profile and all events\n",
" similarity_scores = cosine_similarity(user_vector, event_vectors).flatten()\n",
" \n",
" # Filter out attended events if requested\n",
" if exclude_attended and attended_events:\n",
" for event_id in attended_events:\n",
" event_idx = events[events['event_id'] == event_id].index\n",
" if len(event_idx) > 0:\n",
" similarity_scores[event_idx[0]] = -1 # Set to -1 to exclude from top recommendations\n",
" \n",
" # Get top recommendations\n",
" top_indices = similarity_scores.argsort()[-num_recommendations:][::-1]\n",
" \n",
" # Create results dataframe\n",
" recommended_events = events.iloc[top_indices].copy()\n",
" recommended_events['similarity_score'] = similarity_scores[top_indices]\n",
" \n",
" return recommended_events[['event_id', 'similarity_score']]"
]
},
{
"cell_type": "code",
"execution_count": 18,
"id": "bc343d4e",
"metadata": {},
"outputs": [],
"source": [
"# Hybrid Recommendation System (Collaborative + Content-Based)\n",
"\n",
"def hybrid_recommendation(user_id, num_recommendations=10, collaborative_weight=0.6, content_weight=0.4):\n",
" \"\"\"\n",
" Hybrid recommendation system combining collaborative filtering and content-based filtering\n",
" \n",
" Args:\n",
" user_id: ID of the user to recommend events for\n",
" num_recommendations: Number of events to recommend\n",
" collaborative_weight: Weight for collaborative filtering (0-1)\n",
" content_weight: Weight for content-based filtering (0-1)\n",
" \n",
" Returns:\n",
" DataFrame containing recommended events with combined scores\n",
" \"\"\"\n",
" # Get collaborative filtering recommendations\n",
" collab_events = recommend(user_id) # This returns a set of event_ids\n",
" \n",
" # Get content-based recommendations \n",
" content_recs = content_based_filtering_enhanced(user_id, num_recommendations=20, exclude_attended=True)\n",
" \n",
" # Create a scoring system\n",
" event_scores = {}\n",
" \n",
" # Score content-based recommendations\n",
" for _, row in content_recs.iterrows():\n",
" event_id = row['event_id']\n",
" content_score = row['similarity_score']\n",
" event_scores[event_id] = content_weight * content_score\n",
" \n",
" # Boost scores for collaborative filtering recommendations\n",
" collab_boost = collaborative_weight / len(collab_events) if collab_events else 0\n",
" for event_id in collab_events:\n",
" if event_id in event_scores:\n",
" event_scores[event_id] += collab_boost\n",
" else:\n",
" event_scores[event_id] = collab_boost\n",
" \n",
" # Sort by combined score and get top recommendations\n",
" sorted_events = sorted(event_scores.items(), key=lambda x: x[1], reverse=True)[:num_recommendations]\n",
" \n",
" # Create results dataframe\n",
" recommended_event_ids = [event_id for event_id, _ in sorted_events]\n",
" recommended_scores = [score for _, score in sorted_events]\n",
" \n",
" # Get event details\n",
" result_events = events[events['event_id'].isin(recommended_event_ids)].copy()\n",
" \n",
" # Add scores to the result\n",
" score_dict = dict(sorted_events)\n",
" result_events['hybrid_score'] = result_events['event_id'].map(score_dict)\n",
" result_events = result_events.sort_values('hybrid_score', ascending=False)\n",
" \n",
" return result_events[['event_id']]"
]
},
{
"cell_type": "code",
"execution_count": 19,
"id": "80fb9772",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
" event_id\n",
"226 68e0d79a3048252621b7d89f\n",
"358 68e0d79a3048252621b7d923\n",
"38 68e0d7821f7dc4e22f39d51a\n",
"301 68e0d79a3048252621b7d8ea\n",
"27 68e0d7821f7dc4e22f39d50f\n",
"60 68e0d7821f7dc4e22f39d530\n",
"378 68e0d79a3048252621b7d937\n",
"324 68e0d79a3048252621b7d901\n"
]
}
],
"source": [
"hybrid_recs = hybrid_recommendation('68e0e01c6fbdb8a90b76129b', num_recommendations=8)\n",
"print(hybrid_recs)"
]
}
],
"metadata": {
"kernelspec": {
"display_name": "venv",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.11.14"
}
},
"nbformat": 4,
"nbformat_minor": 5
}