|
| 1 | +"""Measures of central tendency: mean, median, and mode. |
| 2 | +
|
| 3 | +Covers the "Measures of Central Tendency" section of |
| 4 | +https://realpython.com/python-statistics/ |
| 5 | +""" |
| 6 | + |
| 7 | +import math |
| 8 | +import statistics |
| 9 | + |
| 10 | +import numpy as np |
| 11 | +import pandas as pd |
| 12 | +import scipy.stats |
| 13 | + |
| 14 | +from datasets import one_dimensional |
| 15 | + |
| 16 | + |
| 17 | +def mean(): |
| 18 | + """Calculate the arithmetic mean with pure Python, NumPy, and pandas.""" |
| 19 | + x, x_with_nan, y, y_with_nan, z, z_with_nan = one_dimensional() |
| 20 | + |
| 21 | + print("Pure Python:") |
| 22 | + print(f" sum(x) / len(x) = {sum(x) / len(x)}") |
| 23 | + print(f" statistics.mean(x) = {statistics.mean(x)}") |
| 24 | + print(f" statistics.fmean(x) = {statistics.fmean(x)}") |
| 25 | + |
| 26 | + # A single nan in the data makes the result nan. |
| 27 | + print(f" statistics.mean(x_with_nan) = {statistics.mean(x_with_nan)}") |
| 28 | + |
| 29 | + print("NumPy:") |
| 30 | + print(f" np.mean(y) = {np.mean(y)}") |
| 31 | + print(f" y.mean() = {y.mean()}") |
| 32 | + print(f" np.mean(y_with_nan) = {np.mean(y_with_nan)}") |
| 33 | + # Use the nan-safe variant to ignore missing values instead. |
| 34 | + print(f" np.nanmean(y_with_nan) = {np.nanmean(y_with_nan)}") |
| 35 | + |
| 36 | + print("pandas:") |
| 37 | + # pandas skips nan values by default. |
| 38 | + print(f" z.mean() = {z.mean()}") |
| 39 | + print(f" z_with_nan.mean() = {z_with_nan.mean()}") |
| 40 | + |
| 41 | + |
| 42 | +def weighted_mean(): |
| 43 | + """Calculate the weighted mean, where each item has its own weight.""" |
| 44 | + x = [8.0, 1, 2.5, 4, 28.0] |
| 45 | + w = [0.1, 0.2, 0.3, 0.25, 0.15] |
| 46 | + |
| 47 | + wmean = sum(w_ * x_ for (x_, w_) in zip(x, w)) / sum(w) |
| 48 | + print(f"Pure Python: {wmean}") |
| 49 | + |
| 50 | + y, z, w_array = np.array(x), pd.Series(x), np.array(w) |
| 51 | + print(f"np.average(y, weights=w) = {np.average(y, weights=w_array)}") |
| 52 | + print(f"np.average(z, weights=w) = {np.average(z, weights=w_array)}") |
| 53 | + print(f"(w * y).sum() / w.sum() = {(w_array * y).sum() / w_array.sum()}") |
| 54 | + |
| 55 | + |
| 56 | +def harmonic_mean(): |
| 57 | + """Calculate the harmonic mean, the reciprocal of the mean reciprocal.""" |
| 58 | + x, _, y, _, z, _ = one_dimensional() |
| 59 | + |
| 60 | + hmean = len(x) / sum(1 / item for item in x) |
| 61 | + print(f"Pure Python: {hmean}") |
| 62 | + print(f"statistics.harmonic_mean(x): {statistics.harmonic_mean(x)}") |
| 63 | + print(f"scipy.stats.hmean(y): {scipy.stats.hmean(y)}") |
| 64 | + print(f"scipy.stats.hmean(z): {scipy.stats.hmean(z)}") |
| 65 | + |
| 66 | + # A negative value has no harmonic mean. |
| 67 | + try: |
| 68 | + statistics.harmonic_mean([1, 2, -2]) |
| 69 | + except statistics.StatisticsError as error: |
| 70 | + print(f"harmonic_mean([1, 2, -2]) raises StatisticsError: {error}") |
| 71 | + |
| 72 | + |
| 73 | +def geometric_mean(): |
| 74 | + """Calculate the geometric mean, the nth root of the product.""" |
| 75 | + x, _, y, _, z, _ = one_dimensional() |
| 76 | + |
| 77 | + gmean = 1 |
| 78 | + for item in x: |
| 79 | + gmean *= item |
| 80 | + gmean **= 1 / len(x) |
| 81 | + |
| 82 | + print(f"Pure Python: {gmean}") |
| 83 | + print(f"statistics.geometric_mean(x): {statistics.geometric_mean(x)}") |
| 84 | + print(f"scipy.stats.gmean(y): {scipy.stats.gmean(y)}") |
| 85 | + print(f"scipy.stats.gmean(z): {scipy.stats.gmean(z)}") |
| 86 | + |
| 87 | + |
| 88 | +def median(): |
| 89 | + """Find the median, the middle value of the sorted data.""" |
| 90 | + x, x_with_nan, y, y_with_nan, z, z_with_nan = one_dimensional() |
| 91 | + |
| 92 | + n = len(x) |
| 93 | + if n % 2: |
| 94 | + median_ = sorted(x)[round(0.5 * (n - 1))] |
| 95 | + else: |
| 96 | + x_ord, index = sorted(x), round(0.5 * n) |
| 97 | + median_ = 0.5 * (x_ord[index - 1] + x_ord[index]) |
| 98 | + print(f"Pure Python: {median_}") |
| 99 | + |
| 100 | + print(f"statistics.median(x): {statistics.median(x)}") |
| 101 | + # With an even number of items, the median averages the two middle values. |
| 102 | + print(f"median(x[:-1]): {statistics.median(x[:-1])}") |
| 103 | + print(f"median_low(x[:-1]): {statistics.median_low(x[:-1])}") |
| 104 | + print(f"median_high(x[:-1]): {statistics.median_high(x[:-1])}") |
| 105 | + |
| 106 | + print(f"np.median(y): {np.median(y)}") |
| 107 | + print(f"np.nanmedian(y_with_nan): {np.nanmedian(y_with_nan)}") |
| 108 | + print(f"z.median(): {z.median()}") |
| 109 | + print(f"z_with_nan.median(): {z_with_nan.median()}") |
| 110 | + |
| 111 | + |
| 112 | +def mode(): |
| 113 | + """Find the mode, the value that appears most often.""" |
| 114 | + u = [2, 3, 2, 8, 12] |
| 115 | + v = [12, 15, 12, 15, 21, 15, 12] |
| 116 | + |
| 117 | + mode_ = max((u.count(item), item) for item in set(u))[1] |
| 118 | + print(f"Pure Python: {mode_}") |
| 119 | + print(f"statistics.mode(u): {statistics.mode(u)}") |
| 120 | + print(f"statistics.multimode(u): {statistics.multimode(u)}") |
| 121 | + |
| 122 | + # multimode() returns every modal value, mode() only the first. |
| 123 | + print(f"statistics.mode(v): {statistics.mode(v)}") |
| 124 | + print(f"statistics.multimode(v): {statistics.multimode(v)}") |
| 125 | + |
| 126 | + u_array, v_array = np.array(u), np.array(v) |
| 127 | + print(f"scipy.stats.mode(u): {scipy.stats.mode(u_array)}") |
| 128 | + |
| 129 | + result = scipy.stats.mode(v_array) |
| 130 | + print(f"scipy.stats.mode(v): {result}") |
| 131 | + print(f" .mode = {result.mode}") |
| 132 | + print(f" .count = {result.count}") |
| 133 | + |
| 134 | + # pandas returns a Series, so it can report several modal values at once. |
| 135 | + u_series = pd.Series(u) |
| 136 | + v_series = pd.Series(v) |
| 137 | + w_series = pd.Series([2, 2, math.nan]) |
| 138 | + print(f"u.mode():\n{u_series.mode()}") |
| 139 | + print(f"v.mode():\n{v_series.mode()}") |
| 140 | + print(f"w.mode():\n{w_series.mode()}") |
| 141 | + |
| 142 | + |
| 143 | +if __name__ == "__main__": |
| 144 | + for section in ( |
| 145 | + mean, |
| 146 | + weighted_mean, |
| 147 | + harmonic_mean, |
| 148 | + geometric_mean, |
| 149 | + median, |
| 150 | + mode, |
| 151 | + ): |
| 152 | + title = section.__name__.replace("_", " ").title() |
| 153 | + print(f"\n{title}\n{'-' * len(title)}") |
| 154 | + section() |
0 commit comments