SciPy Cheatsheet
Statistics
Use this SciPy reference while you build software engineering projects, review code, or refresh the syntax you reach for most.
Core Import
from scipy import stats import numpy as np # Common direct imports from scipy.stats import ( norm, t, chi2, f, binom, poisson, uniform, expon, beta, gamma, ttest_ind, ttest_rel, ttest_1samp, mannwhitneyu, wilcoxon, kruskal, friedmanchisquare, pearsonr, spearmanr, kendalltau, pointbiserialr, chi2_contingency, fisher_exact, ks_2samp, kstest, anderson, shapiro, normaltest, describe, sem, iqr, zscore, mstats, linregress, theilslopes )
Continuous Distributions
Common Distributions
| Distribution | Class | Key Params | Notes |
|---|---|---|---|
| Normal | norm | loc=μ, scale=σ | |
| Student's t | t | df, loc, scale | |
| Chi-squared | chi2 | df, loc, scale | |
| F | f | dfn, dfd, loc, scale | |
| Exponential | expon | loc, scale=1/λ | scale = mean |
| Beta | beta | a, b, loc, scale | supported on [loc, loc+scale] |
| Gamma | gamma | a, loc, scale | scale = θ = 1/rate |
| Log-normal | lognorm | s=σ, loc, scale=e^μ | |
| Uniform | uniform | loc=a, scale=b-a | on [a, b] |
| Cauchy | cauchy | loc, scale | |
| Pareto | pareto | b, loc, scale | |
| Weibull | weibull_min | c, loc, scale | c = shape |
Distribution Methods
from scipy.stats import norm dist = norm(loc=5, scale=2) # "frozen" distribution (μ=5, σ=2) # PDF, CDF, SF, PPF dist.pdf(6.0) # probability density at x=6 dist.cdf(7.0) # P(X <= 7) dist.sf(7.0) # 1 - CDF = P(X > 7) (more accurate in tails) dist.ppf(0.975) # quantile (inverse CDF) dist.isf(0.025) # inverse SF = ppf(1 - q) # Moments dist.mean() # 5.0 dist.var() # 4.0 dist.std() # 2.0 dist.median() # 5.0 dist.moment(3) # 3rd raw moment dist.stats(moments='mvsk') # mean, variance, skewness, kurtosis # Interval lo, hi = dist.interval(0.95) # 95% coverage interval # Log versions (numerically stable) dist.logpdf(6.0) dist.logcdf(7.0) dist.logsf(7.0) # Random samples samples = dist.rvs(size=1000, random_state=42)
Without Freezing
from scipy.stats import norm norm.pdf(x=6.0, loc=5, scale=2) norm.cdf(x=7.0, loc=5, scale=2) norm.ppf(q=0.975, loc=5, scale=2)
Fitting
from scipy.stats import norm, expon, gamma data = np.random.normal(loc=5, scale=2, size=500) # MLE fit loc, scale = norm.fit(data) a, loc, scale = gamma.fit(data, floc=0) # fix loc=0 # Method of moments (manual) mu, sigma = np.mean(data), np.std(data)
Discrete Distributions
from scipy.stats import binom, poisson, geom, nbinom, hypergeom, bernoulli # Binomial: n=10 trials, p=0.3 b = binom(n=10, p=0.3) b.pmf(3) # P(X=3) b.cdf(4) # P(X<=4) b.ppf(0.95) # 95th percentile b.rvs(100) # 100 samples # Poisson: λ=4 p = poisson(mu=4) p.pmf(3), p.cdf(5) # Negative Binomial nb = nbinom(n=5, p=0.4) # n successes, prob p # Hypergeometric # Drawing k from M with N "successes" hg = hypergeom(M=20, n=7, N=12)
Descriptive Statistics
from scipy.stats import describe, sem, iqr, zscore, gmean, hmean, trim_mean # Summary n, (mn, mx), mean, var, skew, kurt = describe(data) # Single stats sem(data) # standard error of mean iqr(data) # interquartile range iqr(data, rng=(25, 75)) # custom percentile range gmean(data) # geometric mean hmean(data) # harmonic mean trim_mean(data, proportiontocut=0.1) # 10% trimmed mean # Z-scores z = zscore(data) # (x - mean) / std z = zscore(data, axis=0) # along axis z = zscore(data, ddof=1) # with ddof
Hypothesis Tests — Parametric
One-Sample Tests
from scipy.stats import ttest_1samp, chisquare # t-test: is mean == popmean? stat, p = ttest_1samp(data, popmean=5.0) stat, p = ttest_1samp(data, popmean=5.0, alternative='greater') # one-sided # Chi-squared goodness of fit observed = [10, 20, 30, 25, 15] expected = [20, 20, 20, 20, 20] stat, p = chisquare(observed, f_exp=expected)
Two-Sample Tests
from scipy.stats import ttest_ind, ttest_rel, f_oneway, bartlett, levene a = np.random.normal(0, 1, 50) b = np.random.normal(0.5, 1, 50) c = np.random.normal(1.0, 1, 50) # Independent t-test stat, p = ttest_ind(a, b) stat, p = ttest_ind(a, b, equal_var=False) # Welch's t-test stat, p = ttest_ind(a, b, alternative='less') # one-sided # Paired t-test stat, p = ttest_rel(a, b) # One-way ANOVA stat, p = f_oneway(a, b, c) # Variance tests stat, p = bartlett(a, b, c) # equal variances (assumes normality) stat, p = levene(a, b, c) # equal variances (more robust)
Correlation Tests
from scipy.stats import pearsonr, spearmanr, kendalltau, pointbiserialr r, p = pearsonr(x, y) # linear correlation r, p = spearmanr(x, y) # rank correlation tau, p = kendalltau(x, y) # Kendall's tau r, p = pointbiserialr(x_binary, y_continuous) # With confidence interval (SciPy 1.9+) result = pearsonr(x, y) lo, hi = result.confidence_interval(confidence_level=0.95)
Contingency Tables
from scipy.stats import chi2_contingency, fisher_exact, barnard_exact table = np.array([[10, 20], [30, 40]]) # Chi-squared test stat, p, dof, expected = chi2_contingency(table) stat, p, dof, expected = chi2_contingency(table, correction=False) # no Yates' # Fisher's exact (2x2, small samples) oddsratio, p = fisher_exact(table, alternative='two-sided') # Barnard's exact (2x2, no marginal constraint) result = barnard_exact(table)
Hypothesis Tests — Nonparametric
from scipy.stats import ( mannwhitneyu, wilcoxon, kruskal, friedmanchisquare, rankdata, ranksums, mood, ansari, fligner ) a, b, c = (np.random.normal(m, 1, 40) for m in (0, 0.5, 1)) # Mann-Whitney U (independent samples, non-normal) stat, p = mannwhitneyu(a, b, alternative='two-sided', method='auto') # Wilcoxon signed-rank (paired, non-normal) stat, p = wilcoxon(a - b) # differences stat, p = wilcoxon(a, b, alternative='greater') # Kruskal-Wallis (k-sample analog of ANOVA) stat, p = kruskal(a, b, c) # Friedman (repeated measures, non-parametric) stat, p = friedmanchisquare(block1, block2, block3) # Rank sums stat, p = ranksums(a, b)
Normality Tests
from scipy.stats import shapiro, normaltest, anderson, kstest, ks_2samp # (Lilliefors test lives in statsmodels, not SciPy) # Shapiro-Wilk (best for n < 5000) stat, p = shapiro(data) # D'Agostino-Pearson (skewness + kurtosis) stat, p = normaltest(data) # Anderson-Darling (returns critical values for different significance levels) result = anderson(data, dist='norm') print(result.statistic, result.critical_values, result.significance_level) # Kolmogorov-Smirnov stat, p = kstest(data, 'norm', args=(mean, std)) # vs theoretical stat, p = ks_2samp(sample1, sample2) # two-sample KS
Linear Regression
from scipy.stats import linregress, theilslopes, siegelslopes slope, intercept, r, p, se = linregress(x, y) # r = Pearson r; se = standard error of slope # Robust regression (Theil-Sen estimator) result = theilslopes(y, x, alpha=0.95) slope, intercept, lo_slope, hi_slope = result # Siegel's repeated medians (more robust) result = siegelslopes(y, x)
Multiple Testing Correction
from scipy.stats import false_discovery_control p_values = [0.01, 0.04, 0.23, 0.001, 0.05] # Benjamini-Hochberg FDR correction adjusted = false_discovery_control(p_values, method='bh') # Bonferroni: manual bonferroni_alpha = 0.05 / len(p_values)
Probability Utilities
from scipy.stats import chi2, norm # Critical value lookup chi2.ppf(0.95, df=3) # chi-squared critical value at 95%, df=3 t.ppf(0.975, df=29) # two-tailed t critical value # P-value from test statistic 2 * (1 - norm.cdf(abs(z))) # two-tailed z-test p-value chi2.sf(stat, df=3) # right-tail p-value
Kernel Density Estimation
from scipy.stats import gaussian_kde data = np.random.normal(0, 1, 300) kde = gaussian_kde(data) kde = gaussian_kde(data, bw_method='silverman') # 'scott', 'silverman', or float x = np.linspace(-4, 4, 200) density = kde(x) # evaluate KDE kde.integrate_gaussian(0, 1) # integrate KDE against Gaussian kde.set_bandwidth(bw_method=0.3) # change bandwidth
Common Gotchas
ttest_indvs Welch's t-test: Defaultequal_var=Trueassumes equal variances. Useequal_var=False(Welch's) unless you have strong evidence of equal variances — it is more robust.
Normality tests have low power for small n.
shapiroon n=20 will almost never reject. Combine with a Q-Q plot for small samples.
describekurtosis is excess kurtosis (normal = 0), not regular kurtosis (normal = 3).
Frozen vs unfrozen distributions:
norm(loc=5, scale=2).rvs(100)(frozen, fast for repeated use) vsnorm.rvs(loc=5, scale=2, size=100)(unfrozen, fine for one-off). Both are equivalent.
mannwhitneyudefault changed in SciPy 1.7: the default is nowalternative='two-sided'. Before 1.7 the default (alternative=None) returned a one-sided p-value based on the smaller U statistic and emitted a deprecation warning. Always specifyalternativeexplicitly when comparing against old results.