# Laboratorium 5 - rekomendacje grupowe

## Przygotowanie

 * pobierz i wypakuj dataset: https://files.grouplens.org/datasets/movielens/ml-latest-small.zip
   * więcej możesz poczytać tutaj: https://grouplens.org/datasets/movielens/
 * [opcjonalnie] Utwórz wirtualne środowisko
 `python3 -m venv ./recsyslab5`
 * zainstaluj potrzebne biblioteki:
 `pip install numpy pandas scipy matplotlib`

## Część 1. - przygotowanie danych

In [27]:
# importujemy wszystkie potrzebne pakiety

import math
import numpy as np
import pandas as pd
from scipy.sparse.linalg import svds


from random import choice, sample
from statistics import mean, stdev

In [2]:
PATH = 'ml-latest-small'

In [3]:
# wczytujemy oceny uzytkownikow i obliczamy (za pomoc dekompozycji macierzy) wszystkie przewidywane oceny filmow

def read_ratings(path, k=600, scale_factor=2.0, print_stats=True):
    # idea: https://www.kaggle.com/code/indralin/movielens-project-1-2-collaborative-filtering
    reviews = pd.read_csv(f'{path}/ratings.csv', names=['userId', 'movieId', 'rating', 'time'], delimiter=',', engine='python', skiprows=1)

    reviews.drop(['time'], axis=1, inplace=True)
    reviews_no, _ = reviews.shape
    reviews_matrix = reviews.pivot(index='userId', columns='movieId', values='rating')
    movies = reviews_matrix.columns
    users = reviews_matrix.index
    users_no, movies_no = reviews_matrix.shape
    print(f'Got {reviews_no} reviews for {movies_no} movies and {users_no} users.')

    user_ratings_mean = np.nanmean(reviews_matrix.values, axis=1)
    normalized_reviews_matrix = np.nan_to_num(reviews_matrix.values - user_ratings_mean.reshape(-1, 1), 0.0)

    U, sigma, Vt = svds(normalized_reviews_matrix, k=k)
    sigma = np.diag(sigma)
    predicted_ratings = np.dot(np.dot(U, sigma), Vt) + user_ratings_mean.reshape(-1, 1).clip(0.5, 5.0)
    mean_square_error = np.nanmean(np.square(predicted_ratings - reviews_matrix.values))
    std_square_error = np.nanstd(np.square(predicted_ratings - reviews_matrix.values))
    print(f'Reviews prediction mean square error = {mean_square_error}')
    print(f'Reviews prediction standatd deviation of square error = {std_square_error}')

    if print_stats:
        stats = [
            ('metric', 'dataset', 'prediction'),
            ('avg', np.nanmean(reviews_matrix), np.mean(predicted_ratings)),
            ('st_dev', np.nanstd(reviews_matrix), np.std(predicted_ratings)),
            ('median', np.nanmedian(reviews_matrix), np.median(predicted_ratings)),
            ('p25', np.nanquantile(reviews_matrix, 0.25), np.quantile(predicted_ratings, 0.25)),
            ('p75', np.nanquantile(reviews_matrix, 0.75), np.quantile(predicted_ratings, 0.75))
        ]
        print('Stats (for raings in original range [0.5, 5.0]):')
        print('\n'.join([str(s) for s in stats]))

    rounded_predictions = np.rint(scale_factor * predicted_ratings) # cast values to {1, 2, ..., 10}
    return pd.DataFrame(data=rounded_predictions, index=list(users), columns=list(movies))

ratings = read_ratings(PATH)
# dostep do danych:
# ratings[movieId][userId] pobiera 1 wartosc
# ratings.loc[:, movieId] pobiera wektor dla danego filmu
# ratings.loc[userId, :] pobiera wektor dla danego uzytkownika
ratings

Got 100836 reviews for 9724 movies and 610 users.
Reviews prediction mean square error = 1.6577787842924512e-05
Reviews prediction standatd deviation of square error = 0.0007928950536518488
Stats (for raings in original range [0.5, 5.0]):
('metric', 'dataset', 'prediction')
('avg', np.float64(3.501556983616962), np.float64(3.657222337747399))
('st_dev', np.float64(1.042524069618056), np.float64(0.49546237560971024))
('median', np.float64(3.5), np.float64(3.705224008811769))
('p25', np.float64(3.0), np.float64(3.3574517183164017))
('p75', np.float64(4.0), np.float64(3.999981626883002))


Unnamed: 0,1,2,3,4,5,6,7,8,9,10,...,193565,193567,193571,193573,193579,193581,193583,193585,193587,193609
1,8.0,9.0,8.0,9.0,9.0,8.0,9.0,9.0,9.0,9.0,...,9.0,9.0,9.0,9.0,9.0,9.0,9.0,9.0,9.0,9.0
2,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0,...,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0
3,5.0,5.0,5.0,5.0,5.0,5.0,5.0,5.0,5.0,5.0,...,5.0,5.0,5.0,5.0,5.0,5.0,5.0,5.0,5.0,5.0
4,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,...,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0
5,8.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,...,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0
...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...
606,5.0,7.0,7.0,7.0,7.0,7.0,5.0,7.0,7.0,7.0,...,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0
607,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0,...,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0,8.0
608,5.0,4.0,4.0,6.0,6.0,6.0,6.0,6.0,6.0,8.0,...,6.0,6.0,6.0,6.0,6.0,6.0,6.0,6.0,6.0,6.0
609,6.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,8.0,...,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0,7.0


In [5]:
# wczytujemy nazwy filmow i kategorie

movies_metadata = pd.read_csv('ml-latest-small/movies.csv').set_index('movieId')
movies_metadata

Unnamed: 0_level_0,title,genres
movieId,Unnamed: 1_level_1,Unnamed: 2_level_1
1,Toy Story (1995),Adventure|Animation|Children|Comedy|Fantasy
2,Jumanji (1995),Adventure|Children|Fantasy
3,Grumpier Old Men (1995),Comedy|Romance
4,Waiting to Exhale (1995),Comedy|Drama|Romance
5,Father of the Bride Part II (1995),Comedy
...,...,...
193581,Black Butler: Book of the Atlantic (2017),Action|Animation|Comedy|Fantasy
193583,No Game No Life: Zero (2017),Animation|Comedy|Fantasy
193585,Flint (2017),Drama
193587,Bungo Stray Dogs: Dead Apple (2018),Action|Animation


In [6]:
# wczytujemy przykladowe grupy uzytkownikow
groups = pd.read_csv('groups.csv').values.tolist()
groups

[[111, 307, 474, 599, 414],
 [469, 182, 232, 448, 600],
 [508, 581, 497, 402, 566],
 [300, 515, 245, 568, 507],
 [2, 371, 252, 518, 37],
 [269, 360, 469, 287, 308],
 [243, 527, 418, 118, 370],
 [186, 559, 327, 553, 314]]

In [7]:
# przygotowujemy funkcje pomocnicza

def describe_group(group, N=10):
    print(f'\n\nUser ids: {group}')
    group_size = len(group)

    mean_stdev = ratings.loc[group].std(axis=0).mean()
    median_stdev = ratings.loc[group].std(axis=0).median()
    std_stdev = ratings.loc[group].std(axis=0).std()
    print(f'\nMean ratings deviation: {mean_stdev}')
    print(f'Median ratings deviation: {median_stdev}')
    print(f'Standard deviation of ratings deviation: {std_stdev}')

    average_scores = ratings.iloc[group].mean(axis=0)
    average_scores = average_scores.sort_values()
    best_movies = [(movies_metadata['title'][movie_id], average_scores[movie_id]) for movie_id in list(average_scores[-N:].index)]
    worst_movies = [(movies_metadata['title'][movie_id], average_scores[movie_id]) for movie_id in list(average_scores[:N].index)]

    print('\nBest movies:')
    for movie, score in best_movies[::-1]:
        print(f'{movie}, {score}*')
    print('\nWorst movies:')
    for movie, score in worst_movies:
        print(f'{movie}, {score}*')

describe_group(groups[5])



User ids: [269, 360, 469, 287, 308]

Mean ratings deviation: 1.1259149574579788
Median ratings deviation: 1.0954451150103321
Standard deviation of ratings deviation: 0.17836724055768716

Best movies:
Toy Story (1995), 8.2*
Forrest Gump (1994), 8.2*
Willy Wonka & the Chocolate Factory (1971), 8.0*
Braveheart (1995), 8.0*
Terminator 2: Judgment Day (1991), 7.8*
Schindler's List (1993), 7.8*
Dances with Wolves (1990), 7.6*
James and the Giant Peach (1996), 7.6*
Dead Man Walking (1995), 7.6*
Nixon (1995), 7.6*

Worst movies:
Broken Arrow (1996), 5.2*
Sleepy Hollow (1999), 5.4*
The Devil's Advocate (1997), 5.4*
Cable Guy, The (1996), 5.4*
Mission: Impossible (1996), 5.6*
Nutty Professor, The (1996), 5.6*
Down Periscope (1996), 5.8*
Cheech and Chong's Up in Smoke (1978), 5.8*
Wrong Man, The (1956), 5.8*
Fog, The (2005), 5.8*


## Część 2. - algorytmy proste

In [8]:
# zdefiniujmy interfejs dla wszystkich algorytmow rekomendacyjnych

class Recommender:
    def recommend(self, movies, ratings, group, size):
        pass

class RandomRecommender(Recommender):
    def __init__(self):
        self.name = 'random'

    def recommend(self, movies, ratings, group, size):
        return sample(movies, size)

In [9]:
# algorytm rekomendujacy filmy o najwyzszej sredniej ocen

class AverageRecommender(Recommender):
    def __init__(self):
        self.name = 'average'

    def recommend(self, movies, ratings, group, size):
        # Tylko rzędy w których są uzytkownicy z naszej grupy
        selected_rows = ratings.loc[ratings.index.isin(group)]
        # Dla kazdego filmu (kolumny) biezemy średnią ocen z wybranych rzędów
        movies_averages = dict()
        for column in selected_rows.columns:
            movies_averages[column] = selected_rows[column].mean()
        # wyciągamy nazwę filmu z par klucz:wartość - nazwa filmu: średnia ocena
        best_movies = [score[0] for score in sorted(movies_averages.items(), key=lambda x: x[1], reverse=True)[:size]]
        return best_movies

In [10]:
# algorytm rekomendujacy filmy o najwyzszej sredniej ocen,
#   ale rownoczesnie wykluczajacy te filmy, ktore otrzymaly choc jedna ocene ponizej thresholdu

class AverageWithoutMiseryRecommender(Recommender):
    def __init__(self, score_threshold):
        self.name = 'average_without_misery'
        self.score_threshold = score_threshold

    def recommend(self, movies, ratings, group, size):
        selected_rows = ratings.loc[ratings.index.isin(group)]
        movies_averages = dict()
        for column in selected_rows.columns:
            # nie bierzemy pod uwagę recenzji jeśli którykolwiek uzytkownik z grupy bardzo nie lubi
            if not ((selected_rows[column] < self.score_threshold).any()):
                movies_averages[column] = selected_rows[column].mean()
        best_movies = [score[0] for score in sorted(movies_averages.items(), key=lambda x: x[1], reverse=True)[:size]]
        return best_movies

In [11]:
# algorytm uwzgledniajacy preferencje tylko jednego uzytkownika w kazdej iteracji

class FairnessRecommender(Recommender):
    def __init__(self):
        self.name = 'fairness'

    def recommend(self, movies, ratings, group, size):
        selected_rows = ratings.loc[ratings.index.isin(group)]
        best_movies = []
        for i in range(size):
            user = group[i % len(group)]
            column = self.find_best_movie(user, size, best_movies, selected_rows)
            best_movies.append(column)
        return best_movies

    def find_best_movie(self, user, size, best_movies, selected_rows):
        # size+1 najlepiej ocenionych przez uzytkownika filmów
        max_films = selected_rows.loc[user].nlargest(size+1).index
        for film in max_films:
            if film not in best_movies:
                return film

        return None


In [12]:
# wybrany algorytm wyborczy (dyktatura, Borda, Copeland)

class VotingRecommender(Recommender):
    def __init__(self):
        self.name = "dictator"

    # dyktatura
    def recommend(self, movies, ratings, group, size):
        selected_rows = ratings.loc[ratings.index.isin(group)]
        random = np.random.randint(0, len(group))
        return selected_rows.loc[group[random]].nlargest(size).index


In [13]:
# algorytm zachlanny, aproksymujacy metode Proportional Approval Voting
#   w kazdej iteracji wybieramy ten film, ktory najbardziej zwieksza zadowolenie zgodnie z punktacja PAV

class ProportionalApprovalVotingRecommender(Recommender):
    def __init__(self, threshold):
        self.threshold = threshold
        self.name = 'PAV'

    def recommend(self, movies, ratings, group, size):

        # kazdy element ma wagę 1/i, zalenie od pozycji i na ktorej stoi
        movies_scores = {movie: 0 for movie in movies}
        for user_id in group:
            # po kolei filmy dla kazdego z uczestników, z wagami opisanymi powyzej
            for i, movie_id in enumerate(ratings.loc[user_id].sort_values(ascending=False).index):
                movies_scores[movie_id] += 1 / (i + 1)


        movie_scores = [(movies_scores[movie], movie) for movie in movies]
        movie_scores.sort()
        return [movie_pair[1] for movie_pair in movie_scores[:size]]

## Część 3. - funkcje celu

In [14]:
# dwie funkcje pomocnicze:
#  - znajdujaca ulubione filmy danego uzytkownika
#  - obliczajaca sume ocen wystawionych przez uzytkownika wszystkim filmom w rekomendacji

def top_n_movies_for_user(ratings, movies, user_id, n):
    return ratings.loc[user_id].nlargest(n).index

def total_score(recommendation, user_id, ratings):
    return ratings.loc[user_id, recommendation].sum()

In [25]:
# funkcja obliczajaca zadowolenie pojedynczego uzytkownika
#  - iloraz zadowolenia z wygenerowanej rekomendacji oraz zadowolenia z hipotetycznej rekomendacji idealnej
def overall_user_satisfaction(recommendation, user_id, movies, ratings):
    recommendation_score = total_score(recommendation, user_id, ratings)
    best_score = total_score(top_n_movies_for_user(ratings, movies, user_id, len(recommendation)), user_id, ratings)
    if best_score == 0:
        return 1
    else:
        return recommendation_score / best_score

# funkcja celu - srednia z zadowolenia wszystkich uzytkownikow w grupie
def overall_group_satisfaction(recommendation, group, movies, ratings):
    score = 0
    for user in group:
        score += overall_user_satisfaction(recommendation, user, movies, ratings)
    if len(group) == 0:
        return 0
    return score / len(group)

# funkcja celu - roznica miedzy maksymalnym i minimalnym zadowolenie w grupie
def group_disagreement(recommendation, group, movies, ratings):
    if len(group) == 0:
        return 0
    maximum = 0
    minimum = 10
    for user in group:
        score = overall_user_satisfaction(recommendation, user, movies, ratings)
        maximum = max(score, maximum)
        minimum=min(score,minimum)
    return maximum - minimum

## Część 4. - Sequential Hybrid Aggregation

In [16]:
# algorytm balansujacy pomiedzy wyborem elementow o najwyzszej sredniej ocen
#   i o najwyzszej minimalnej ocenie
#   wyliczajacy w kazdej iteracji parametr alfa - jak na wykladzie
class SequentialHybridAggregationRecommender(Recommender):
     def __init__(self):
        self.name = 'sequential_hybrid_aggregation'

     def recommend(self, movies, ratings, group, size):
        selected_rows = ratings.loc[ratings.index.isin(group)]
        alfa = 0
        best_movies = []
        for _ in range(size):
            movies_scores = dict()
            for movie in [col for col in selected_rows.columns if col not in best_movies]:
                movies_scores[movie] = (1 - alfa) * selected_rows[movie].mean() + alfa * selected_rows[movie].min()
            best_movie = sorted(movies_scores.items(),key=lambda x: x[1], reverse=True)[0][0]
            best_movies.append(best_movie)
            alfa = group_disagreement(best_movies, group, movies, ratings)
        return best_movies

## Część 5. - porównanie algorytmów

In [28]:
recommenders = [
    RandomRecommender(),
    AverageRecommender(),
    AverageWithoutMiseryRecommender(5),
    FairnessRecommender(),
    VotingRecommender(), # dyktator
    ProportionalApprovalVotingRecommender(5), # liczymy PAV
    SequentialHybridAggregationRecommender()
]

recommendation_size = 10

# dla kazdego algorytmu:
#  - wygenerujmy jedna rekomendacje dla kazdej grupy
#  - obliczmy wartosci obu funkcji celu dla kazdej rekomendacji
#  - obliczmy srednia i odchylenie standardowe dla obu funkcji celu

movies = ratings.columns.tolist()

for recommender in recommenders:
    print(f'{recommender.name}')
    group_satisfaction = []
    group_disagreements = []
    for group in groups:
        recommendation = recommender.recommend(movies, ratings, group, recommendation_size)
        group_satisfaction.append(overall_group_satisfaction(recommendation, group, movies, ratings))
        group_disagreements.append(group_disagreement(recommendation, group, movies, ratings))
    print(f'Group satisfaction: {mean(group_satisfaction)} +- {stdev(group_satisfaction)}')
    print(f'Group disagreement: {mean(group_disagreements)} +- {stdev(group_disagreements)}')

random
Group satisfaction: 0.7398138337280458 +- 0.094714995740798
Group disagreement: 0.21971044204935958 +- 0.05719058496886814
average
Group satisfaction: 0.8750686713670882 +- 0.042366915974012695
Group disagreement: 0.19084232731554157 +- 0.0637420765471257
average_without_misery
Group satisfaction: 0.8701906644942359 +- 0.04652105241846132
Group disagreement: 0.20250899398220823 +- 0.07465871412049092
fairness
Group satisfaction: 0.8200296082476679 +- 0.0675113876802403
Group disagreement: 0.1721614990107258 +- 0.06907354057283835
dictator
Group satisfaction: 0.8087087119229976 +- 0.06251777599327504
Group disagreement: 0.31870580808080806 +- 0.08564515504967421
PAV
Group satisfaction: 0.7110638337280458 +- 0.13822463092526371
Group disagreement: 0.2547104420493596 +- 0.07040894622577887
sequential_hybrid_aggregation
Group satisfaction: 0.8736223091975117 +- 0.04196602800751999
Group disagreement: 0.1962766548927263 +- 0.026423750945568248
