R uses a uniform d/p/q/r naming convention:
| Prefix | Meaning | Example (Normal) |
|---|---|---|
d | Density / PMF | dnorm(0) = 0.3989 |
p | CDF P(X ≤ q) | pnorm(1.96) = 0.975 |
q | Quantile (inverse CDF) | qnorm(0.975) = 1.96 |
r | Random sample | rnorm(10) |
Replace norm with any other distribution name: binom, pois, t, chisq, f, exp, unif, gamma, beta, geom, nbinom, hyper, etc.
# Density at x = 1.5 for N(0, 1)
dnorm(1.5) # 0.1295
# Density at x for N(60, 10)
dnorm(70, mean = 60, sd = 10)
# P(X <= 1.96)
pnorm(1.96) # 0.975
# P(50 <= X <= 70) for N(60, 10)
pnorm(70, 60, 10) - pnorm(50, 60, 10) # 0.6827
# 95th percentile for N(0,1)
qnorm(0.95) # 1.645
# Random sample of 1000 from N(70, 5)
set.seed(123)
x <- rnorm(1000, mean = 70, sd = 5)
hist(x, breaks = 30, col = "lightblue", probability = TRUE,
main = "1000 samples ~ N(70, 5)")
curve(dnorm(x, 70, 5), add = TRUE, col = "red", lwd = 2)
d gives the curve height, p the area to the left of q, q goes backwards from a probability to the cut-off, and r draws random values. The same picture applies to every distribution family.# P(X = 3) for B(10, 0.5)
dbinom(3, size = 10, prob = 0.5) # 0.117
# P(X <= 3)
pbinom(3, 10, 0.5) # 0.172
# 95th percentile (smallest k with P(X ≤ k) ≥ 0.95)
qbinom(0.95, 10, 0.5) # 8
# Random sample
rbinom(15, size = 10, prob = 0.4) # 15 simulated counts
# P(X = 5) when λ = 3
dpois(5, lambda = 3) # 0.1008
# P(X <= 2)
ppois(2, 3) # 0.4232
# 90th percentile
qpois(0.90, 3) # 5
# Random sample
rpois(100, lambda = 4)
dexp(2, rate = 0.5) # Exponential
punif(0.7) # Uniform(0,1)
dt(2, df = 10) # Student's t
pchisq(11, df = 5) # Chi-square
qf(0.95, df1 = 4, df2 = 20) # F-distribution
# Marks ~ N(60, 15)
# P(marks > 80)
1 - pnorm(80, mean = 60, sd = 15) # 0.0912
# P(45 < marks < 75)
pnorm(75, 60, 15) - pnorm(45, 60, 15) # 0.683
# i.e., 68.3% of students fall within ±1σ — empirical rule
# A coin tossed 20 times. P(at most 8 heads) =
pbinom(8, 20, 0.5) # 0.2517
# Always set seed first for reproducibility
set.seed(42)
# Sample from N(0,1)
rnorm(5)
# Sample of size 5 from 1..100 without replacement
sample(1:100, 5)
# Sample with replacement
sample(1:6, 10, replace = TRUE) # 10 die rolls
# Weighted sampling
sample(c("A","B","C"), 50, replace = TRUE, prob = c(0.5, 0.3, 0.2))
set.seed(1)
N <- 5000 # number of samples
n <- 30 # sample size
mu <- 50; sigma <- 10
means <- replicate(N, mean(rnorm(n, mu, sigma)))
hist(means, breaks = 30, freq = FALSE, col = "skyblue",
main = "Sampling Distribution of the Mean (n = 30)",
xlab = expression(bar(X)))
curve(dnorm(x, mu, sigma/sqrt(n)), add = TRUE, col = "red", lwd = 2)
# The histogram tracks the theoretical N(50, 10/sqrt(30)) curve — CLT verified.
data <- c(48, 52, 49, 53, 51, 47, 55, 50, 49, 52)
t.test(data, mu = 50) # H0: mu = 50
p > 0.05 ⇒ fail to reject H₀; the mean is consistent with 50. The 95 % CI (48.84, 52.36) contains 50.
group1 <- c(72, 75, 78, 80, 76)
group2 <- c(65, 68, 70, 72, 67)
# Default: Welch's t-test (unequal variances)
t.test(group1, group2)
# Pooled (equal variances) — set var.equal = TRUE
t.test(group1, group2, var.equal = TRUE)
# One-tailed
t.test(group1, group2, alternative = "greater")
before <- c(72, 78, 69, 80, 85, 76, 82, 74)
after <- c(70, 76, 67, 78, 83, 74, 79, 72)
t.test(before, after, paired = TRUE)
# Equivalent to: t.test(before - after)
setosa <- iris$Sepal.Length[iris$Species == "setosa"]
versicolor <- iris$Sepal.Length[iris$Species == "versicolor"]
t.test(setosa, versicolor) # p < 2.2e-16 ⇒ reject H0
data(sleep) # 10 subjects, 2 drugs
t.test(extra ~ group, data = sleep, paired = TRUE)
# Die rolled 60 times, observed face counts
observed <- c(8, 11, 9, 12, 10, 10)
chisq.test(observed) # default: equal expected probabilities
p ≈ 0.96 ⇒ fail to reject H₀; the die appears fair.
With explicit expected probabilities (Mendelian 9:3:3:1):
observed <- c(460, 140, 130, 70)
chisq.test(observed, p = c(9, 3, 3, 1)/16)
tbl <- matrix(c(60, 40,
20, 80), nrow = 2, byrow = TRUE,
dimnames = list(c("Smoker","Non-smoker"),
c("Cancer","No Cancer")))
chisq.test(tbl)
p < 0.001 ⇒ reject independence; smoking and cancer are associated.
For small expected counts (any cell < 5), use Fisher's Exact Test: fisher.test(tbl).
# H0: sigma^2 = sigma0^2; manual calculation
n <- 25
s2 <- 18
sigma0_2 <- 16
chi2 <- (n - 1) * s2 / sigma0_2
chi2 # 27
# Two-tailed p-value
2 * min(pchisq(chi2, n - 1), 1 - pchisq(chi2, n - 1))
# Iris: Sepal.Length differs across 3 Species?
fit <- aov(Sepal.Length ~ Species, data = iris)
summary(fit)
p < 0.001 ⇒ at least one species mean differs.
TukeyHSD(fit)
All three pairwise differences are significant.
data(ToothGrowth) # built-in
fit2 <- aov(len ~ supp * dose, data = ToothGrowth) # main + interaction
summary(fit2)
# Without interaction
aov(len ~ supp + factor(dose), data = ToothGrowth) |> summary()
var.test(group1, group2) # equivalent to F-test
# Or
var.test(extra ~ group, data = sleep)
fit <- aov(mpg ~ factor(cyl), data = mtcars)
summary(fit) # F = 39.7, p < 0.001 ⇒ cyl matters
TukeyHSD(fit)
data(ToothGrowth)
fit <- aov(len ~ supp * factor(dose), data = ToothGrowth)
summary(fit)
# main effects of supp & dose AND their interaction tested in one shot
All major R test functions return a list containing statistic, p.value, conf.int, estimate etc., which you can extract programmatically:
res <- t.test(group1, group2)
res$statistic # t value
res$parameter # degrees of freedom
res$p.value # p-value
res$conf.int # 95 % CI for the difference of means
res$estimate # sample means
You can also compute CIs manually:
# 95 % CI for the mean
x <- rnorm(50, 100, 15)
m <- mean(x); se <- sd(x)/sqrt(length(x))
m + c(-1, 1) * qt(0.975, df = length(x) - 1) * se
Use when normality assumption of t-test is violated.
# One-sample
data <- c(48, 52, 49, 53, 51, 47, 55, 50, 49, 52)
wilcox.test(data, mu = 50) # H0: median = 50
# Paired (Wilcoxon signed-rank)
wilcox.test(before, after, paired = TRUE)
wilcox.test(group1, group2)
# Equivalent to Mann-Whitney U test
wilcox.test(extra ~ group, data = sleep)
# Both on the same data — does the conclusion agree?
t.test(group1, group2)$p.value
wilcox.test(group1, group2)$p.value
before <- c(72, 78, 69, 80, 85, 76, 82, 74)
after <- c(70, 76, 67, 78, 83, 74, 79, 72)
t.test(before, after, paired = TRUE)$p.value # tiny
wilcox.test(before, after, paired = TRUE)$p.value # also tiny — both reject H0
income_men <- c(20, 22, 28, 30, 35, 50, 80, 120)
income_women <- c(18, 19, 25, 27, 30, 40, 60, 90)
shapiro.test(income_men)$p.value # < 0.05 ⇒ normality doubtful
# So prefer Wilcoxon over t-test
wilcox.test(income_men, income_women)
shapiro.test(), Q-Q plot.var.test() or bartlett.test().# Example workflow
shapiro.test(group1); shapiro.test(group2)
var.test(group1, group2)
if (both_p > 0.05) {
t.test(group1, group2, var.equal = TRUE)
} else {
wilcox.test(group1, group2)
}
d/p/q/r prefixes for any distribution (Normal, Binomial, Poisson, t, χ², F, …).t.test() handles one-sample, two-sample (Welch or pooled) and paired tests.chisq.test() for goodness of fit and independence; fisher.test() when expected counts are small.aov() + summary() for ANOVA; follow up with TukeyHSD().wilcox.test() for non-parametric one-sample, paired and two-sample tests.shapiro.test() and var.test().