In [7]:
import numpy as np
import pandas as pd
from scipy.signal import correlate
import matplotlib.pyplot as plt
from scipy.sparse import lil_matrix
from itertools import product
from itertools import combinations
from datetime import datetime, timedelta
from multiprocessing import Pool


# Load the temporal series data
data_with_date = pd.read_csv('temporal_series_6.csv')
data = data_with_date.drop(columns=['date'])

#Inizialize parameters
max_tau = 15 
threshold = 0.05

#Function to compute maximum of time correlation function between two temporal series
def time_delayed_cross_correlation(df, col1, col2, max_tau, threshold=0.05):
    """
    Optimized custom time-delayed cross-correlation function for two time series.

    Args:
    df (pandas.DataFrame): Dataframe containing the time series data.
    col1 (str): Name of the first time series column.
    col2 (str): Name of the second time series column.
    max_tau (int): Maximum time delay (positive or negative).
    threshold (float): Threshold for maximum correlation.

    Returns:
    float: Maximum cross-correlation above the threshold, or 0 if none.
    """
    # Extract the series and compute means and std deviations
    series_i = df[col1].values
    series_j = df[col2].values
    mean_i, std_i = series_i.mean(), series_i.std()
    mean_j, std_j = series_j.mean(), series_j.std()
    
    # Check if standard deviations are zero to avoid division by zero
    if std_i == 0 or std_j == 0:
        return 0
    
    # Initialize array to store cross-correlations for each lag
    cross_corrs = np.zeros(2 * max_tau + 1)

    # Compute cross-correlation for each lag
    for tau in range(-max_tau, max_tau + 1):
        if tau < 0:
            # Positive lag: shift series_j forward by -tau (series_i aligns with delayed series_j)
            numerator = np.sum((series_i[-tau:] - mean_i) * (series_j[:len(series_j) + tau] - mean_j))
            denominator = (std_i * std_j * (len(series_i) + tau))
        else:
            # Negative or zero lag: shift series_i forward by tau (series_j aligns with delayed series_i)
            numerator = np.sum((series_i[:len(series_i) - tau] - mean_i) * (series_j[tau:] - mean_j))
            denominator = (std_i * std_j * (len(series_i) - tau))

        # Calculate cross-correlation, check for zero denominator
        cross_corrs[tau + max_tau] = numerator / denominator if denominator != 0 else 0

    # Find the maximum correlation and compare with threshold
    max_corr = np.max(cross_corrs)
    max_index = np.argmax(cross_corrs)
    best_tau = max_index - max_tau  # Convert array index back to tau value

    # Apply threshold check and adjust max_corr based on tau direction
    if max_corr >= threshold:
        return max_corr if best_tau >= 0 else -max_corr
    else:
        return 0


#Function to compute the whole adjacency matrix (impossible to compute)
def ccr_matrix(df):
    #build the adjacency matrix
    adjacency_matrix = pd.DataFrame(np.nan, index=df.columns, columns=df.columns)

    #iterate for each pair of columns
    for col1 in df.columns:
        for col2 in df.columns:
            adjacency_matrix.loc[col1,col2] = time_delayed_cross_correlation(df, col1, col2, max_tau)

    return adjacency_matrix

#function to compute just a submatrix of the adjacency matrix (submatricese on the diagonal)
def compute_submatrix(df, columns, filename, max_tau):
    submatrix = pd.DataFrame(index=columns, columns=columns)

    # Loop through column pairs, skipping redundant calculations
    for col1, col2 in product(columns, repeat=2):
        if col1 <= col2:  # Ensures each pair is calculated only once
            correlation = time_delayed_cross_correlation(df, col1, col2, max_tau)
            submatrix.loc[col1, col2] = correlation
            submatrix.loc[col2, col1] = correlation  # Fill symmetric position

    submatrix.to_csv(filename)
    print(f"Submatrix {filename} has been saved.")
    return submatrix

#function to computer a submatrix which is not on the diagonal
def compute_cross_group_matrix(df, group_a, group_b, filename):
    cross_matrix = pd.DataFrame(index=group_a, columns=group_b)
    for col1, col2 in product(group_a, group_b):
        cross_matrix.loc[col1, col2] = time_delayed_cross_correlation(df, col1, col2, max_tau)
    cross_matrix.to_csv(filename)
    print(f"submatrix{filename} has been saved")
    return cross_matrix

#Now to compute the submatrices we need to divide the 3104 columns into groups
def divide_into_groups(df, group_sizes):
    #Divides the comlumns into specified sizes
    columns = df.columns.tolist()
    groups = []
    start = 0
    for size in group_sizes:
        groups.append(columns[start:start + size])
        start += size
    return groups

#Change of program, now we will try to compute manually the single pieces
group_1 = data.columns[0:1000]
group_2 = data.columns[1000:2000]
group_3 = data.columns[2000:3104]


#filename_sub = "6_sub_mat_1.csv"
#compute_submatrix(data, group_1, filename_sub, max_tau)

filename_cross = "6_cross_matrix_2_3.csv"
compute_cross_group_matrix(data, group_2, group_3, filename_cross)


submatrix6_cross_matrix_1_3.csv has been saved


Unnamed: 0,38099.0,38101.0,38103.0,38105.0,39001.0,39003.0,39005.0,39007.0,39009.0,39011.0,...,56027.0,56029.0,56031.0,56033.0,56035.0,56037.0,56039.0,56041.0,56043.0,56045.0
1001.0,-0.441169,-0.446006,-0.313054,-0.406079,-0.555558,-0.582849,-0.600266,-0.645056,-0.51818,-0.521466,...,-0.55497,-0.433901,-0.371569,-0.547067,-0.465114,-0.543668,-0.569881,-0.556461,-0.623181,-0.224733
1003.0,-0.359121,-0.38705,0.165543,-0.246218,-0.509038,-0.484369,-0.4552,-0.543508,-0.474256,-0.451887,...,-0.47696,-0.243875,-0.312693,-0.507153,-0.474222,-0.445895,-0.48457,-0.524832,-0.47903,-0.164453
1005.0,0.220991,0.199133,0.284432,0.318659,-0.15639,-0.142258,0.198136,-0.090975,-0.191994,0.186265,...,-0.145129,0.131905,0.149506,-0.171541,-0.156589,-0.126557,-0.19337,-0.14115,-0.224075,0.344084
1007.0,-0.444584,-0.442172,-0.233249,-0.298181,-0.499564,0.569952,0.445824,-0.493491,0.44561,-0.537901,...,-0.399077,-0.31078,-0.398543,-0.474632,-0.444607,0.565805,-0.515857,-0.470868,0.61255,-0.175874
1009.0,-0.502396,-0.497107,0.35609,0.382559,-0.622764,-0.700719,-0.636431,-0.715874,-0.484043,0.609769,...,-0.591102,-0.527811,-0.451534,-0.627922,-0.512688,-0.618011,-0.632761,-0.644701,-0.582669,-0.30851
...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...
21073.0,-0.590591,-0.643546,-0.209891,-0.300149,-0.647298,0.672755,0.592419,-0.713527,0.582907,-0.621984,...,-0.658113,-0.463088,-0.425838,-0.622137,-0.574304,-0.603136,-0.614139,-0.675299,-0.622012,-0.528834
21075.0,0.156028,-0.135607,0.227447,0.209829,-0.22135,0.197928,0.190336,-0.22113,0.325931,0.194913,...,-0.195422,0.262428,0.229857,-0.190358,-0.257187,0.248332,-0.248117,-0.180887,0.238618,-0.237418
21077.0,-0.51323,-0.50409,-0.410181,-0.494744,-0.607838,-0.700343,-0.707982,-0.65494,0.577832,-0.556244,...,-0.566523,-0.318832,-0.326741,-0.619489,-0.645907,-0.587274,-0.621512,-0.685372,-0.667536,-0.170686
21079.0,-0.604869,-0.574557,0.433413,0.332607,-0.673396,0.663342,0.604913,-0.698248,0.631483,0.62829,...,-0.610674,0.504947,-0.522917,-0.678764,-0.546537,-0.586035,-0.632464,-0.638731,0.650283,-0.355927
