
num_friends = [100.0,49,41,40,25,21,21,19,19,18,18,16,15,15,15,15,14,14,13,13,13,13,12,12,11,10,10,10,10,10,10,10,10,10,10,10,10,10,10,10,9,9,9,9,9,9,9,9,9,9,9,9,9,9,9,9,9,9,8,8,8,8,8,8,8,8,8,8,8,8,8,7,7,7,7,7,7,7,7,7,7,7,7,7,7,7,6,6,6,6,6,6,6,6,6,6,6,6,6,6,6,6,6,6,6,6,6,6,5,5,5,5,5,5,5,5,5,5,5,5,5,5,5,5,5,4,4,4,4,4,4,4,4,4,4,4,4,4,4,4,4,4,4,4,4,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,2,2,2,2,2,2,2,2,2,2,2,2,2,2,2,2,2,1,1,1,1,1,1,1,1,1,1,1,1,1,1,1,1,1,1,1,1,1,1]

print(num_friends)

from collections import Counter
import matplotlib.pyplot as plt

friend_counts = Counter(num_friends)

xs = range(101)                         # largest value is 100
ys = [friend_counts[x] for x in xs]     # height is just # of friends

plt.bar(xs, ys)
plt.axis([0, 101, 0, 25])
plt.title("Histogram of Friend Counts")
plt.xlabel("# of friends")
plt.ylabel("# of people")

# plt.show()

# **************************************************
# Some common statistics:

# Cardinality of the data set
num_points = len(num_friends)               	# 204

# Max and min value in data set
largest_value = max(num_friends) 		# 100
smallest_value = min(num_friends) 		# 1

# *********************************************************
# Ranking

sorted_values = sorted(num_friends)

smallest_value = sorted_values[0] 		# 1
second_smallest_value = sorted_values[1] 	# 1

largest_value  = sorted_values[-1] 		# 100
second_largest_value = sorted_values[-2] 	# 49


# *********************************************************
# Mean:

def mean(x):
    return sum(x) / len(x)

print(mean(num_friends))		 # 7.333333

# ***********************************************************
# Median:
#
#   If we have five data points in a sorted vector x, the median is x[5 // 2] or x[2]. 
#   If we have six data points, we want the average of x[2] (2 = 6//2-1) and
#   and x[3] (3 - 6//2)

def median(v):
    """finds the 'middle-most' value of v"""

    n = len(v)
    sorted_v = sorted(v)
    midpoint = n // 2

    if n % 2 == 1:
        # if odd, return the middle value
        return sorted_v[midpoint]
    else:
        # if even, return the average of the middle values
        lo = midpoint - 1
        hi = midpoint
        return (sorted_v[lo] + sorted_v[hi]) / 2

print(median(num_friends))			# 6.0


# ***********************************************************
# Quantile:

def quantile(x, p):
    """returns the pth-percentile value in x"""

    p_index = int(p * len(x))
    return sorted(x)[p_index]

# **************************************************************
# Mode: (most frequent element)


def mode(x):
    """returns a list, might be more than one mode"""

    counts = Counter(x)			# Format: [item1:4, item2:6, ...]  (dict)

    max_count = max(counts.values())	# 6

    return [x_i for x_i, count in counts.items()
            if count == max_count]


print( mode(num_friends) )




# @@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@
# Dispersion refers to **measures of how spread out** the data is. 
#
# Small dispersion == data are not spread out much
# Large dispersion == data are not spread out a lot
# @@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@



# ******************************************************************
# range = max value - min value
#

def data_range(x):		# We can't use range because it's a Python function
    return max(x) - min(x)

print(data_range(num_friends))	 	# 99


# ******************************************************************
# variance =  sum_all_x_i ( (x_i - x_mean)^2 ) / (n-1)
#

# This function takes a data set (vector) and returns a
# "normalized" data set with mean = 0

def de_mean(x):
    """
    translate x by subtracting its mean (so the result has mean 0)
    """
    x_bar = mean(x)
    return [x_i - x_bar for x_i in x]


# Variance

from scratch.linear_algebra import dot
from scratch.linear_algebra import sum_of_squares

#def dot(v, w):
#    return sum(v_i * w_i for v_i, w_i in zip(v, w))
#
#def sum_of_squares(v):
#    """v_1 * v_1 + ... + v_n * v_n"""
#    return dot(v,v)

def variance(x):
    """
    assumes x has at least two elements
    """
    n = len(x)
    deviations = de_mean(x)
    return sum_of_squares(deviations) / (n - 1)

# ******************************************************************
# standard deviation = sqrt(variable)
#

import math

def standard_deviation(x):
    return math.sqrt(variance(x))

print(variance(num_friends))		 	# 81.54
print(standard_deviation(num_friends))		# 9,03


# *******************************************************************************
# Inter-quantile range = range between the i-th quantile and the j-th quantile
#
# A common inter-quantile range is:
#
#       Q(75%) - Q(25%)
#


def interquartile_range(x):
    return quantile(x, 0.75) - quantile(x, 0.25)

print(interquartile_range(num_friends))



# @@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@
# Correlation:
#
#     measuring interconnectivity (dependency) between 2 data sets
#
# @@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@

# Example:
#    DataSciencester’s VP of Growth has a theory that the amount of time people 
#    spend on the site is related to the number of friends they have on the site 
#    and she’s asked you to verify this.
#
# After digging through traffic logs, you’ve come up with a list daily_minutes 
# that shows how many minutes per day each user spends on DataSciencester, 
# and you’ve ordered it so that its elements correspond to the elements of 
# our previous "num_friends" list.

daily_minutes = [1,68.77,51.25,52.08,38.36,44.54,57.13,51.4,41.42,31.22,34.76,54.01,38.79,47.59,49.1,27.66,41.03,36.73,48.65,28.12,46.62,35.57,32.98,35,26.07,23.77,39.73,40.57,31.65,31.21,36.32,20.45,21.93,26.02,27.34,23.49,46.94,30.5,33.8,24.23,21.4,27.94,32.24,40.57,25.07,19.42,22.39,18.42,46.96,23.72,26.41,26.97,36.76,40.32,35.02,29.47,30.2,31,38.11,38.18,36.31,21.03,30.86,36.07,28.66,29.08,37.28,15.28,24.17,22.31,30.17,25.53,19.85,35.37,44.6,17.23,13.47,26.33,35.02,32.09,24.81,19.33,28.77,24.26,31.98,25.73,24.86,16.28,34.51,15.23,39.72,40.8,26.06,35.76,34.76,16.13,44.04,18.03,19.65,32.62,35.59,39.43,14.18,35.24,40.13,41.82,35.45,36.07,43.67,24.61,20.9,21.9,18.79,27.61,27.21,26.61,29.77,20.59,27.53,13.82,33.2,25,33.1,36.65,18.63,14.87,22.2,36.81,25.53,24.62,26.25,18.21,28.08,19.42,29.79,32.8,35.99,28.32,27.79,35.88,29.06,36.28,14.1,36.63,37.49,26.9,18.58,38.48,24.48,18.95,33.55,14.24,29.04,32.51,25.63,22.22,19,32.73,15.16,13.9,27.2,32.01,29.27,33,13.74,20.42,27.32,18.23,35.35,28.48,9.08,24.62,20.12,35.26,19.92,31.02,16.49,12.16,30.7,31.22,34.65,13.13,27.51,33.2,31.57,14.1,33.42,17.44,10.12,24.42,9.82,23.39,30.93,15.03,21.67,31.09,33.29,22.61,26.89,23.48,8.38,27.81,32.35,23.84]

daily_hours = [dm / 60 for dm in daily_minutes]

# Here's the scatter plot

plt.clf()				# Clear previous figure
plt.scatter(num_friends, daily_minutes)      
# plt.show()


# %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%
# Co-variance: the paired analogue of variance
#
#    Whereas variance measures how a single variable deviates from its mean, 
#    covariance measures how two variables vary in tandem from their means
#

def covariance(x, y):
    n = len(x)
    return dot(de_mean(x), de_mean(y)) / (n - 1)

# Comment:
#
#   Covariance ranges between -infinity and infinity
#
#   (1) a “large” **positive** covariance (~= 1) means that 
#       x tends to be large when y is large and small when y is small. 
#   (2) a “large” **negative** covariance means the opposite: 
#       x tends to be small (= neg) when y is large (= pos) and vice versa. 
#   (3) a covariance close to zero means that no relationship exists
#       between x and y values... (i.e.: no co-related)

print(covariance(num_friends, daily_minutes)) 		# 22.43

# %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%
# Correlation:
#
#    Co-variance is hard to interpret, we commonly use the correlation
#    to measure dependency between 2 sets of data
#

def correlation(x, y):
    stdev_x = standard_deviation(x)
    stdev_y = standard_deviation(y)

    if stdev_x > 0 and stdev_y > 0:
        return covariance(x, y) / (stdev_x * stdev_y)
    else:
        return 0 			# if variation = 0, correlation is zero

# Comment:
#
#   correlation is unitless and always lies between 
#
#       -1 (perfect anti-correlation) and 
#        1 (perfect correlation).

print(correlation(num_friends, daily_minutes))		 # 0.25

# 0.25 represents a relatively weak positive correlation.


# !!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!
# Important fact:
#
#    correlation can be very sensitive to outliers.
#
# And there is a single outlier in the data:
#
#      The person with 100 friends spent only 1 min per day on the site !!!

# Re-examine correlation without the outliers:

outlier = num_friends.index(100) # returns the position at the first occurrence 
				 # of a specified value. 

num_friends_good = [x for i, x in enumerate(num_friends) if i != outlier] 
daily_minutes_good = [x for i, x in enumerate(daily_minutes) if i != outlier]

print(correlation(num_friends_good, daily_minutes_good))



x = [-2, -1, 0, 1, 2]
y = [ 2, 1, 0, 1, 2]
print(correlation(x,y))


x = [-2, 1, 0, 1, 2]
y = [99.98, 99.99, 100, 100.01, 100.02]
print(correlation(x,y))
