A matrix is a two-dimensional collection of elements of the same mode,
arranged in rows and columns. Build it with matrix(), filling by column by default.
M <- matrix(1:6, nrow = 2, ncol = 3) # fills by column
M <- matrix(1:6, nrow = 2, byrow = TRUE) # fill by row instead
dim(M) # 2 3
M[1, 2] # element row 1, col 2
M[ , 2] # entire 2nd column
M[1, ] # entire 1st row
A <- matrix(1:4, 2, 2)
A <- rbind(A, c(5, 6)) # add a row
A <- cbind(A, c(7, 8, 9)) # add a column
A <- A[-1, ] # remove the 1st row
B <- matrix(c(1,2,3,4), 2, 2)
t(B) # transpose
B %*% B # matrix multiplication
solve(B) # inverse (B must be square, non-singular)
det(B) # determinant
B + B; 3 * B # element-wise add / scalar multiply
rowSums(B); colMeans(B)
M <- matrix(c(0,12,13,8,20, 12,0,15,28,88, 13,15,0,6,9,
8,28,6,0,33, 20,88,9,33,0), nrow = 5, byrow = TRUE)
rownames(M) <- colnames(M) <- paste0("C", 1:5)
# shortest non-zero distance:
min(M[M > 0]) # 6
which(M == 6, arr.ind = TRUE) # rows/cols of the pair (C3, C4)
Finds the pair of cities with the shortest distance (C3–C4, distance 6).
# prices in 2 shops (rows = items, cols = shops)
price <- matrix(c(1.5,1, 2,2.5, 5,4.5, 16,17), nrow = 4, byrow = TRUE)
# demand (rows = persons, cols = items)
demand <- matrix(c(6,5,3,1, 3,6,3,2, 3,4,3,1), nrow = 3, byrow = TRUE)
cost <- demand %*% price # 3x2: each person's total in each shop
cost
apply(cost, 1, which.min) # cheapest shop for each person
A data frame is a table where each column can be a different mode (numeric, character, factor) but all columns have the same length. It is the standard structure for real datasets — rows = observations, columns = variables.
df <- data.frame(
name = c("Asha","Ravi","Sita"),
marks = c(78, 65, 88),
pass = c(TRUE, TRUE, TRUE)
)
str(df) # structure of the data frame
nrow(df); ncol(df); names(df)
df$marks # a column by name
df[ , "name"] # same column by index
df[df$marks > 70, ] # rows where marks > 70 (filtering)
df$grade <- ifelse(df$marks >= 75, "A", "B") # add a column
df <- df[ , -3] # remove the 3rd column
df <- df[-2, ] # remove the 2nd row
head(df); tail(df) # first / last rows
summary(df) # per-column summary
apply(df[ ,c("marks")], 2, mean) # apply a function over columns
aggregate(marks ~ grade, data = df, FUN = mean) # group means
students <- data.frame(id = 1:3, name = c("A","B","C"))
scores <- data.frame(id = c(1,2,3), marks = c(60, 75, 90))
merged <- merge(students, scores, by = "id") # join on id
marks <- data.frame(
section = c("A","A","A","B","B","B"),
student = c(1,2,3,1,2,3),
M1 = c(46,34,56,43,67,76),
M2 = c(54,55,66,44,76,68),
M3 = c(45,55,64,45,78,37)
)
marks$total <- marks$M1 + marks$M2 + marks$M3 # row totals
marks # display
aggregate(total ~ section, marks, max) # highest total per section
marks$M4 <- c(50, 60, 70, 55, 65, 40) # new column filled with marks
marks$total <- marks$M1 + marks$M2 + marks$M3 + marks$M4 # recompute
colMeans(marks[ , c("M1","M2","M3","M4")]) # subject averages
A factor stores categorical data as a set of levels (e.g. "Male"/"Female"). A table tabulates the frequencies of factor levels (a contingency table).
gender <- factor(c("M","F","F","M","M","F"))
levels(gender) # "F" "M"
table(gender) # F:3 M:3
# two-way table
region <- factor(c("N","S","N","S","N","S"))
table(gender, region) # cross-tabulation
prop.table(table(gender)) # proportions
table(c("Pass","Fail","Pass","Pass","Fail")) returns Fail = 2, Pass = 3 — an instant
frequency distribution of a categorical variable.
Ordered factor for grades: grade <- factor(c("B","A","C"), levels = c("C","B","A"), ordered = TRUE);
then grade[1] < grade[2] returns TRUE (B < A) because the order is defined.
Exploratory Data Analysis is the first stage of any analysis: summarising the main features of a dataset — centre, spread, shape, missing values and outliers — usually with numbers and simple plots, before formal modelling.
x <- c(12, 15, 14, 10, 18, 20, 13, 16)
mean(x) # arithmetic mean
median(x) # median
range(x) # min and max -> 10 20
diff(range(x)) # numeric range -> 10
var(x) # sample variance
sd(x) # sample standard deviation
quantile(x) # quartiles
# mode (R has no built-in) :
mode_fn <- function(v){ u <- unique(v); u[which.max(tabulate(match(v,u)))] }
summary(x)
# Min. 1st Qu. Median Mean 3rd Qu. Max.
# 10.00 12.75 14.50 14.75 16.50 20.00
summary() on a data frame reports these five-number-summary statistics (plus the mean) for
every numeric column, and counts for factors — the quickest one-line overview of a dataset.
# missing values
y <- c(3, NA, 7, 9, NA, 12)
mean(y, na.rm = TRUE) # ignore NAs
y[is.na(y)] <- mean(y, na.rm = TRUE) # impute with the mean
# outliers via the IQR rule
Q <- quantile(x, c(0.25, 0.75)); IQR <- Q[2] - Q[1]
low <- Q[1] - 1.5*IQR; high <- Q[2] + 1.5*IQR
x[x < low | x > high] # values flagged as outliers
# min-max normalization
norm01 <- function(v) (v - min(v)) / (max(v) - min(v))
norm01(x)
# z-score standardization
scale(x) # built-in: subtracts mean, divides by sd
data <- c(46, 54, 45, 34, 55, 64, 78, 90, 41, 66)
summary(data) # five-number summary + mean
sd(data) # spread
boxplot(data) # visual check for outliers
v <- c(20, NA, 35, 50, NA, 80)
v[is.na(v)] <- median(v, na.rm = TRUE) # impute with median (robust)
v_scaled <- (v - min(v)) / (max(v) - min(v)) # min-max to [0,1]
round(v_scaled, 3)
R offers base graphics (built-in) and the powerful ggplot2
package (grammar of graphics). Other packages: lattice, plotly (interactive),
scatterplot3d and rgl (3-D).
x <- c(5, 8, 12, 6, 9)
barplot(x, names.arg = c("A","B","C","D","E"), main = "Bar Chart")
hist(rnorm(100), main = "Histogram", col = "skyblue")
boxplot(x, main = "Box Plot")
plot(1:5, x, type = "l", main = "Line Chart") # line
plot(1:5, x, main = "Scatter Plot") # scatter (points)
pie(x, labels = c("A","B","C","D","E")) # pie
library(ggplot2)
df <- data.frame(x = 1:5, y = c(5,8,12,6,9))
ggplot(df, aes(x = x, y = y)) +
geom_line(colour = "steelblue") +
geom_point(size = 3) +
labs(title = "ggplot2 line + points", x = "Index", y = "Value") +
theme_minimal()
# scatterplot3d package
library(scatterplot3d)
scatterplot3d(x = mtcars$wt, y = mtcars$hp, z = mtcars$mpg,
main = "3D Scatter", color = "darkgreen")
# base persp() for a surface
persp(z = volcano, theta = 30, phi = 25, col = "lightblue")
marks <- c(46,54,45,34,55,64,78,90,41,66,72,58,49,61,80)
hist(marks, breaks = 6, col = "lightblue",
main = "Distribution of Marks", xlab = "Marks")
study <- c(2,4,6,8,10); score <- c(15,23,28,36,43)
plot(study, score, pch = 19, main = "Study hours vs Score")
abline(lm(score ~ study), col = "red") # least-squares regression line
The lm() function fits the linear regression (Learning Outcome 5) and
abline() overlays it.
t(), %*%, solve(), det().$, filter with logical conditions, combine with merge().table() builds frequency / contingency tables.summary() gives a one-line overview.na.rm / imputation; flag outliers with the 1.5×IQR rule; normalize by min-max or z-score (scale()).ggplot2; 3-D via scatterplot3d / persp(); fit lines with lm() + abline().