# Laboratorium 5 - rekomendacje grupowe

## Przygotowanie

 * pobierz i wypakuj dataset: https://files.grouplens.org/datasets/movielens/ml-latest-small.zip
   * więcej możesz poczytać tutaj: https://grouplens.org/datasets/movielens/
 * [opcjonalnie] Utwórz wirtualne środowisko
 `python3 -m venv ./recsyslab5`
 * zainstaluj potrzebne biblioteki:
 `pip install numpy pandas scipy matplotlib`

## Część 1. - przygotowanie danych

In [2]:
# importujemy wszystkie potrzebne pakiety

import math
import numpy as np
import pandas as pd
from scipy.sparse.linalg import svds


from random import choice, sample
from statistics import mean, stdev

In [3]:
PATH = 'ml-latest-small'

In [4]:
# wczytujemy oceny uzytkownikow i obliczamy (za pomoc dekompozycji macierzy) wszystkie przewidywane oceny filmow

def read_ratings(path, k=600, scale_factor=2.0, print_stats=True):
    # idea: https://www.kaggle.com/code/indralin/movielens-project-1-2-collaborative-filtering
    reviews = pd.read_csv(f'{path}/ratings.csv', names=['userId', 'movieId', 'rating', 'time'], delimiter=',', engine='python', skiprows=1)
    
    reviews.drop(['time'], axis=1, inplace=True)
    reviews_no, _ = reviews.shape
    reviews_matrix = reviews.pivot(index='userId', columns='movieId', values='rating')
    movies = reviews_matrix.columns
    users = reviews_matrix.index
    users_no, movies_no = reviews_matrix.shape
    print(f'Got {reviews_no} reviews for {movies_no} movies and {users_no} users.')

    user_ratings_mean = np.nanmean(reviews_matrix.values, axis=1)
    normalized_reviews_matrix = np.nan_to_num(reviews_matrix.values - user_ratings_mean.reshape(-1, 1), 0.0)

    U, sigma, Vt = svds(normalized_reviews_matrix, k=k)
    sigma = np.diag(sigma)
    predicted_ratings = np.dot(np.dot(U, sigma), Vt) + user_ratings_mean.reshape(-1, 1).clip(0.5, 5.0)
    mean_square_error = np.nanmean(np.square(predicted_ratings - reviews_matrix.values))
    std_square_error = np.nanstd(np.square(predicted_ratings - reviews_matrix.values))
    print(f'Reviews prediction mean square error = {mean_square_error}')
    print(f'Reviews prediction standatd deviation of square error = {std_square_error}')

    if print_stats:
        stats = [
            ('metric', 'dataset', 'prediction'),
            ('avg', np.nanmean(reviews_matrix), np.mean(predicted_ratings)),
            ('st_dev', np.nanstd(reviews_matrix), np.std(predicted_ratings)),
            ('median', np.nanmedian(reviews_matrix), np.median(predicted_ratings)),
            ('p25', np.nanquantile(reviews_matrix, 0.25), np.quantile(predicted_ratings, 0.25)),
            ('p75', np.nanquantile(reviews_matrix, 0.75), np.quantile(predicted_ratings, 0.75))
        ]
        print('Stats (for raings in original range [0.5, 5.0]):')
        print('\n'.join([str(s) for s in stats]))

    rounded_predictions = np.rint(scale_factor * predicted_ratings) # cast values to {1, 2, ..., 10}
    return pd.DataFrame(data=rounded_predictions, index=list(users), columns=list(movies))
    
ratings = read_ratings(PATH)
# dostep do danych:
# ratings[movieId][userId] pobiera 1 wartosc
# ratings.loc[:, movieId] pobiera wektor dla danego filmu
# ratings.loc[userId, :] pobiera wektor dla danego uzytkownika
ratings

Got 100836 reviews for 9724 movies and 610 users.
Reviews prediction mean square error = 1.6577787842924292e-05
Reviews prediction standatd deviation of square error = 0.0007928950536518226
Stats (for raings in original range [0.5, 5.0]):
('metric', 'dataset', 'prediction')
('avg', np.float64(3.501556983616962), np.float64(3.657222337747399))
('st_dev', np.float64(1.042524069618056), np.float64(0.49546237560971024))
('median', np.float64(3.5), np.float64(3.705224008811769))
('p25', np.float64(3.0), np.float64(3.357451718316402))
('p75', np.float64(4.0), np.float64(3.999981626883001))


Unnamed: 0,1,2,3,4,5,6,7,8,9,10,...,193565,193567,193571,193573,193579,193581,193583,193585,193587,193609
1,8.0,9.0,8.0,9.0,9.0,8.0,9.0,9.0,9.0,9.0,...,9.0,9.0,9.0,9.0,9.0,9.0,9.0,9.0,9.0,9.0
2,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0,...,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0
3,5.0,5.0,5.0,5.0,5.0,5.0,5.0,5.0,5.0,5.0,...,5.0,5.0,5.0,5.0,5.0,5.0,5.0,5.0,5.0,5.0
4,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,...,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0
5,8.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,...,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0
...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...
606,5.0,7.0,7.0,7.0,7.0,7.0,5.0,7.0,7.0,7.0,...,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0
607,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0,...,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0
608,5.0,4.0,4.0,6.0,6.0,6.0,6.0,6.0,6.0,8.0,...,6.0,6.0,6.0,6.0,6.0,6.0,6.0,6.0,6.0,6.0
609,6.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,8.0,...,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0


In [5]:
# wczytujemy nazwy filmow i kategorie

movies_metadata = pd.read_csv('ml-latest-small/movies.csv').set_index('movieId')
movies_metadata

Unnamed: 0_level_0,title,genres
movieId,Unnamed: 1_level_1,Unnamed: 2_level_1
1,Toy Story (1995),Adventure|Animation|Children|Comedy|Fantasy
2,Jumanji (1995),Adventure|Children|Fantasy
3,Grumpier Old Men (1995),Comedy|Romance
4,Waiting to Exhale (1995),Comedy|Drama|Romance
5,Father of the Bride Part II (1995),Comedy
...,...,...
193581,Black Butler: Book of the Atlantic (2017),Action|Animation|Comedy|Fantasy
193583,No Game No Life: Zero (2017),Animation|Comedy|Fantasy
193585,Flint (2017),Drama
193587,Bungo Stray Dogs: Dead Apple (2018),Action|Animation


In [7]:
# wczytujemy przykladowe grupy uzytkownikow
groups = pd.read_csv('groups.csv').values.tolist()
groups

[[111, 307, 474, 599, 414],
 [469, 182, 232, 448, 600],
 [508, 581, 497, 402, 566],
 [300, 515, 245, 568, 507],
 [2, 371, 252, 518, 37],
 [269, 360, 469, 287, 308],
 [243, 527, 418, 118, 370],
 [186, 559, 327, 553, 314]]

In [8]:
# przygotowujemy funkcje pomocnicza

def describe_group(group, N=10):
    print(f'\n\nUser ids: {group}')
    group_size = len(group)
    
    mean_stdev = ratings.loc[group].std(axis=0).mean()
    median_stdev = ratings.loc[group].std(axis=0).median()
    std_stdev = ratings.loc[group].std(axis=0).std()
    print(f'\nMean ratings deviation: {mean_stdev}')
    print(f'Median ratings deviation: {median_stdev}')
    print(f'Standard deviation of ratings deviation: {std_stdev}')
    
    average_scores = ratings.iloc[group].mean(axis=0)
    average_scores = average_scores.sort_values()
    best_movies = [(movies_metadata['title'][movie_id], average_scores[movie_id]) for movie_id in list(average_scores[-N:].index)]
    worst_movies = [(movies_metadata['title'][movie_id], average_scores[movie_id]) for movie_id in list(average_scores[:N].index)]
    
    print('\nBest movies:')
    for movie, score in best_movies[::-1]:
        print(f'{movie}, {score}*')
    print('\nWorst movies:')
    for movie, score in worst_movies:
        print(f'{movie}, {score}*')

describe_group(groups[5])



User ids: [269, 360, 469, 287, 308]

Mean ratings deviation: 1.1259149574579788
Median ratings deviation: 1.0954451150103321
Standard deviation of ratings deviation: 0.17836724055768716

Best movies:
Forrest Gump (1994), 8.2*
Toy Story (1995), 8.2*
Braveheart (1995), 8.0*
Willy Wonka & the Chocolate Factory (1971), 8.0*
Terminator 2: Judgment Day (1991), 7.8*
Schindler's List (1993), 7.8*
Nixon (1995), 7.6*
Twelve Monkeys (a.k.a. 12 Monkeys) (1995), 7.6*
Batman Begins (2005), 7.6*
James and the Giant Peach (1996), 7.6*

Worst movies:
Broken Arrow (1996), 5.2*
The Devil's Advocate (1997), 5.4*
Sleepy Hollow (1999), 5.4*
Cable Guy, The (1996), 5.4*
Mission: Impossible (1996), 5.6*
Nutty Professor, The (1996), 5.6*
Matrix Revolutions, The (2003), 5.8*
Happy Gilmore (1996), 5.8*
D3: The Mighty Ducks (1996), 5.8*
Cheech and Chong's Up in Smoke (1978), 5.8*


## Część 2. - algorytmy proste

In [None]:
# zdefiniujmy interfejs dla wszystkich algorytmow rekomendacyjnych

class Recommender:
    def recommend(self, movies: list[int], ratings: pd.DataFrame, group: list[int], size: int) -> list[int]:
        pass


# jako pierwszy zaimplementujemy algorytm losowy - dla porownania
    
class RandomRecommender(Recommender):
    def __init__(self):
        self.name = 'random'
        
    def recommend(self, movies: list[int], ratings: pd.DataFrame, group: list[int], size: int) -> list[int]:
        return sample(movies, size)

In [12]:
# algorytm rekomendujacy filmy o najwyzszej sredniej ocen

class AverageRecommender(Recommender):
    def __init__(self):
        self.name = 'average'
    
    def recommend(self, movies: list[int], ratings: pd.DataFrame, group: list[int], size: int) -> list[int]:
        average_scores = ratings.iloc[group].mean(axis=0)
        return list(average_scores.sort_values(ascending=False).index[:size])

In [11]:
# algorytm rekomendujacy filmy o najwyzszej sredniej ocen,
#   ale rownoczesnie wykluczajacy te filmy, ktore otrzymaly choc jedna ocene ponizej thresholdu

class AverageWithoutMiseryRecommender(Recommender):
    def __init__(self, score_threshold):
        self.name = 'average_without_misery'
        self.score_threshold = score_threshold
        
    def recommend(self, movies: list[int], ratings: pd.DataFrame, group: list[int], size: int) -> list[int]:
        average_scores = ratings.iloc[group].mean(axis=0)
        # filter out movies with at least one rating below the threshold
        average_scores = average_scores[~ratings.iloc[group].lt(self.score_threshold).any()]
        return list(average_scores.sort_values(ascending=False).index[:size])

In [None]:
# algorytm uwzgledniajacy preferencje tylko jednego uzytkownika w kazdej iteracji

class FairnessRecommender(Recommender):
    def __init__(self):
        self.name = 'fairness'
        
    def recommend(self, movies: list[int], ratings: pd.DataFrame, group: list[int], size: int) -> list[int]:
        recommended = []
        for i in range(size):
            user = group[i % len(group)]
            # filter out movies already recommended
            user_ratings = ratings.loc[user]
            user_ratings = user_ratings[~user_ratings.index.isin(recommended)]

            best_movie = user_ratings.idxmax()
            recommended.append(best_movie)
        return recommended

In [29]:
# wybrany algorytm wyborczy (dyktatura, Borda, Copeland)

# Algorytm Bordy
# • Każdy użytkownik przyznaje elementom punkty –
#   od 0 punktów dla elementu najmniej lubianego
#   do N punktów dla elementu najbardziej lubianego
# • Sumujemy liczby punktów dla każdego elementu
# • Wybieramy elementy z największą sumą punktów

class VotingRecommender(Recommender):
    def __init__(self):
        self.name = 'Bord'
    
    def recommend(self, movies, ratings, group, size):
        borda_scores = {movie_id: 0 for movie_id in movies}
        for user in group:
            user_ratings = ratings.loc[user]
            user_ratings = user_ratings.sort_values(ascending=False)

            previous_rating = -1
            previous_points = -1
            N = len(movies)
            for i, (movie_id, rating) in enumerate(user_ratings.items()):
                points = N - i - 1
                borda_scores[movie_id] += points if rating != previous_rating else previous_points
                previous_rating = rating
                previous_points = points
        return [movie_id for movie_id, _ in sorted(borda_scores.items(), key=lambda x: x[1], reverse=True)[:size]]

In [31]:
# algorytm zachlanny, aproksymujacy metode Proportional Approval Voting
#   w kazdej iteracji wybieramy ten film, ktory najbardziej zwieksza zadowolenie zgodnie z punktacja PAV

class ProportionalApprovalVotingRecommender(Recommender):
    def __init__(self, threshold):
        self.threshold = threshold
        self.name = 'PAV'
        
    def recommend(self, movies, ratings, group, size):
        recommended = []
        for i in range(size):
            user = group[i % len(group)]
            # filter out movies already recommended
            user_ratings = ratings.loc[user]
            user_ratings = user_ratings[~user_ratings.index.isin(recommended)]
            
            # calculate PAV score for each movie
            PAV_scores = {movie_id: 0 for movie_id in user_ratings.index}
            for movie_id, rating in user_ratings.items():
                PAV_scores[movie_id] = sum([1 for user_id in group if ratings.loc[user_id, movie_id] >= self.threshold])
            best_movie = max(PAV_scores.items(), key=lambda x: x[1])[0]
            recommended.append(best_movie)
        return recommended

## Część 3. - funkcje celu

In [16]:
# dwie funkcje pomocnicze:
#  - znajdujaca ulubione filmy danego uzytkownika
#  - obliczajaca sume ocen wystawionych przez uzytkownika wszystkim filmom w rekomendacji

def top_n_movies_for_user(ratings, movies, user_id, n):
    user_ratings = ratings.loc[user_id]
    return user_ratings.sort_values(ascending=False).index[:n]

def total_score(recommendation, user_id, ratings):
    return sum(ratings.loc[user_id, recommendation])

In [17]:
# funkcja obliczajaca zadowolenie pojedynczego uzytkownika
#  - iloraz zadowolenia z wygenerowanej rekomendacji oraz zadowolenia z hipotetycznej rekomendacji idealnej
def overall_user_satisfaction(recommendation, user_id, movies, ratings):
    top_movies = top_n_movies_for_user(ratings, movies, user_id, len(recommendation))
    return total_score(recommendation, user_id, ratings) / total_score(top_movies, user_id, ratings)

# funkcja celu - srednia z zadowolenia wszystkich uzytkownikow w grupie
def overall_group_satisfaction(recommendation, group, movies, ratings):
    return mean([overall_user_satisfaction(recommendation, user_id, movies, ratings) for user_id in group])

# funkcja celu - roznica miedzy maksymalnym i minimalnym zadowolenie w grupie
def group_disagreement(recommendation, group, movies, ratings):
    return max([overall_user_satisfaction(recommendation, user_id, movies, ratings) for user_id in group]) - min([overall_user_satisfaction(recommendation, user_id, movies, ratings) for user_id in group])

## Część 4. - Sequential Hybrid Aggregation

In [None]:
# algorytm balansujacy pomiedzy wyborem elementow o najwyzszej sredniej ocen
#   i o najwyzszej minimalnej ocenie
#   wyliczajacy w kazdej iteracji parametr alfa - jak na wykladzie

class SequentialHybridAggregationRecommender(Recommender):
    def __init__(self):
        self.name = 'sequential_hybrid_aggregation'
    
    def recommend(self, movies, ratings, group, size):
        selected_rows = ratings.loc[ratings.index.isin(group)]
        alfa = 0
        best_movies = []
        for _ in range(size):
            movies_scores = dict()
            for movie in [col for col in selected_rows.columns if col not in best_movies]:
                movies_scores[movie] = (1 - alfa) * selected_rows[movie].mean() + alfa * selected_rows[movie].min()
            best_movie = sorted(movies_scores.items(),key=lambda x: x[1], reverse=True)[0][0]
            best_movies.append(best_movie)
            alfa = group_disagreement(best_movies, group, movies, ratings)
        return best_movies

## Część 5. - porównanie algorytmów

In [39]:
recommenders = [
    RandomRecommender(),
    AverageRecommender(),
    AverageWithoutMiseryRecommender(5),
    FairnessRecommender(),
    VotingRecommender(),
    ProportionalApprovalVotingRecommender(5),
    SequentialHybridAggregationRecommender()
]

recommendation_size = 10

# dla kazdego algorytmu:
#  - wygenerujmy jedna rekomendacje dla kazdej grupy
#  - obliczmy wartosci obu funkcji celu dla kazdej rekomendacji
#  - wypiszmy wyniki na konsole

movies = ratings.columns.tolist()
for recommender in recommenders:
    print(f'\n\n{recommender.name}')
    for group in groups:
        recommendation = recommender.recommend(movies, ratings, group, recommendation_size)
        print(f'\nGroup {group}')
        print(f'Recommendation: {recommendation}')
        print(f'Overall group satisfaction: {overall_group_satisfaction(recommendation, group, movies, ratings)}')
        print(f'Group disagreement: {group_disagreement(recommendation, group, movies, ratings)}')



random

Group [111, 307, 474, 599, 414]
Recommendation: [1615, 4993, 59336, 170399, 131749, 46347, 115680, 3915, 25963, 161044]
Overall group satisfaction: 0.642
Group disagreement: 0.22999999999999998

Group [469, 182, 232, 448, 600]
Recommendation: [891, 99910, 1685, 131104, 111785, 386, 44864, 170399, 94160, 4802]
Overall group satisfaction: 0.6599999999999999
Group disagreement: 0.09999999999999998

Group [508, 581, 497, 402, 566]
Recommendation: [193585, 107449, 5401, 881, 3022, 97328, 6320, 4754, 2699, 212]
Overall group satisfaction: 0.7589003436426117
Group disagreement: 0.26116838487972516

Group [300, 515, 245, 568, 507]
Recommendation: [7067, 3177, 1111, 27320, 4005, 8905, 101973, 5240, 173535, 132462]
Overall group satisfaction: 0.8697804576376005
Group disagreement: 0.24242424242424243

Group [2, 371, 252, 518, 37]
Recommendation: [4407, 4229, 96737, 4975, 73323, 33830, 1759, 8197, 112556, 2824]
Overall group satisfaction: 0.8287528344671202
Group disagreement: 0.1222222