abline(v=x.hmean,lwd=3,col="red")
text(5,300,"Harmonic mean",pos=4)
segments(5,300,x.hmean,300,col="red",lwd=2)
abline(v=x.gmean,lwd=3,col="orange")
text(5,330,"Geometric mean",pos=4)
segments(5,330,x.gmean,330,col="orange",lwd=2)
x <- x5
x.mean <- mean(x)
x.hmean <- 1/mean(1/x)
x.gmean <- exp(mean(log(x)))
hist(x,breaks=40,ylim=c(0,400),col="gold")
abline(v=x.mean,lwd=3,col="navy")
text(5,360,"Arithmetic mean",pos=4)
segments(5,360,x.mean,360,col="navy",lwd=2)
abline(v=x.hmean,lwd=3,col="red")
text(5,300,"Harmonic mean",pos=4)
segments(5,300,x.hmean,300,col="red",lwd=2)
abline(v=x.gmean,lwd=3,col="orange")
text(5,330,"Geometric mean",pos=4)
segments(5,330,x.gmean,330,col="orange",lwd=2)
x <- x4
x.mean <- mean(x)
x.hmean <- 1/mean(1/x)
x.gmean <- exp(mean(log(x)))
hist(x,breaks=20,ylim=c(0,1000),col="gold")
abline(v=x.mean,lwd=3,col="navy")
text(20,900,"Arithmetic mean",pos=1)
segments(20,900,x.mean,900,col="navy",lwd=2)
abline(v=x.hmean,lwd=3,col="red",lty=3)
text(20,700,"Harmonic mean",pos=4)
segments(20,700,x.hmean,700,col="red",lwd=2,lty=3)
abline(v=x.gmean,lwd=3,col="orange")
text(20,800,"Geometric mean",pos=4)
segments(20,800,x.gmean,800,col="orange",lwd=2)
x.mean <- 99
hist(x,breaks=20,ylim=c(0,120),col="gold")
abline(v=x.mean,lwd=3,col="blue")
text(x.mean,90,"Arithmetic\nmean",pos=4)
x1 <- (1:20)
x1 <- (1:20)
par(mfrow=c(1,2))
plot(x1, main="before",sub=paste("mean=",round(mean(x1),2),"sd=",round(sd(x1),2)),
cex=.5,col="navy")
x1.norm <- (x1-mean(x1))/sd(x1)
plot(x1.norm, main="after",sub=paste("mean=",round(mean(x1.norm),2),"sd=",round(sd(x1.norm),2)),
cex=.5,col="navy")
mean(x1.norm)  #[1] 0
sd(x1.norm)    #[1] 1
x1 <- (1:20)
par(mfrow=c(1,2))
plot(x1, main="before",sub=paste("mean=",round(mean(x1),2),"sd=",round(sd(x1),2)),
cex=.5,col="navy")
x1.norm <- (x1-mean(x1))/sd(x1)
plot(x1.norm, main="after",sub=paste("mean=",round(mean(x1.norm),2),"sd=",round(sd(x1.norm),2)),
cex=.5,col="navy")
mean(x1.norm)  #[1] 0
sd(x1.norm)    #[1] 1
x2 <- runif(100) + rnorm(100)*3
hist(x2, main="before",sub=paste("mean=",round(mean(x2),2),"sd=",round(sd(x2),2)),cex=.5,col="navy")
x2.norm <- (x2-mean(x2))/sd(x2)
hist(x2.norm, main="after",sub=paste("mean=",round(mean(x2.norm),2),"sd=",round(sd(x2.norm),2)),
cex=.5,col="navy")
mean(x2.norm)  #[1] almost 0
sd(x2.norm)    #[1] 1
x3 <- seq(1,100,3)
plot(x3, main="before",sub=paste("mean=",round(mean(x3),2),"sd=",round(sd(x3),2)),
cex=.5,col="navy")
x3.norm <- (x3-mean(x3))/sd(x3)
plot(x3.norm, main="after",sub=paste("mean=",round(mean(x3.norm),2),"sd=",round(sd(x3.norm),2)),
cex=.5,col="navy")
mean(x3.norm)  #[1] 0
sd(x3.norm)    #[1] 1
x4 <- c(runif(10),300)
hist(x4, main="before",sub=paste("mean=",round(mean(x4),2),"sd=",round(sd(x4),2)),
cex=.5,col="navy")
x4.norm <- (x4-mean(x4))/sd(x4)
hist(x4.norm, main="after",sub=paste("mean=",round(mean(x4.norm),2),"sd=", round(sd(x4.norm),2)),
cex=.5,col="navy")
mean(x4.norm)  #[1] almost 0
sd(x4.norm)    #[1] 1
x5 <- 1/(runif(50)*3)
plot(x5, main="before",sub=paste("mean=",round(mean(x5),2),"sd=",round(sd(x5),2)),
cex=.5,col="navy")
x5.norm <- (x5-mean(x5))/sd(x5)
plot(x5.norm, main="after",sub=paste("mean=",round(mean(x5.norm),2),"sd=",round(sd(x5.norm),2)),
cex=.5,col="navy")
mean(x5.norm)  #[1] almost 0
sd(x5.norm)    #[1] 1
round(3.225262626)
round(3.225262626,2)
round(3.225262626,-2)
round(2535263.225262626,-2)
floor(2535263.225262626,-2)
floor(2535263.225262626)
ceiling(2535263.225262626)
ceiling(2535263.225262626*100)/100
ceiling(2535263.225262626/100)*100
pwd()
wd()
setwd("~/Dropbox/courses/5210-2026/web-5210/psy5210/Projects/Chapter2")
?matplot
matplot
help.search("average")
##Set the column headers after reading in:
data <- read.table("c5data.txt")
colnames(data) <- head
data <- read.table("c5data2.txt",skip=1, header=F)
colnames(data) <- head
##read in a version with the header:
data <- read.table("c5data2.txt",header=T)
data
data
data[1,]
data[10,]
1:10
data[1:10,]
data[c(1,5,9,11,22),]
a  <- data[c(1,5,9,11,22),]
a
data
data[,3]
data[1:5,]
data$step
data[,3:5]
data[,5:8]
data[,c(1,5:8)]
c(1,5,6,7,8)
c(1,5:8)
c(1,5:8,9, 11:15)
data[,c(1,5:8)]
data[,-c(1,5:8)]
plot(x$distoff)
plot(data$distoff)
order(data$distoff)
order(-data$distoff)
data[order(-data$distoff),]
data[1:3,]
data[3:1,]
data[c(15,3,99,5,1),]
order(-data$distoff)
data[order(-data$distoff),]
data[order(-data$distoff),] -> tmp
tmp$targx
plot(tmp$targx)
data[order(-data$distoff),]$targx
data[order(-data$distoff),][1:5,]
data[order(-data$distoff),][1:5,][,3]
data[order(-data$distoff),][1:5,][,3:5]
##create a voter database:
party <-  c("R","R","D","R","R","D","D","D","R","R","D")
gender <- c("M","M","F","F","F","F","M","M","F","M","M")
vote   <- c("A","B","A","A","A","B","A","A","B","B","A")
survey <- data.frame(party,gender,vote)
survey
table(survey$party)
table(survey$party,gender)
table(survey$party,gender,vote)
tapply(survey$party,list(survey$gender),length)
tapply(survey$party,list(survey$gender),mean)
trees
###########################################
##2.5.  Aggregate and tapply functions
###########################################
set.seed(111)
x <- rnorm(500)                  ##generate random numbers
y <- x + runif(500,-.3,.3)       ##related random numbers
dat3 <- data.frame(x=x,y=y)      ##create a data frame
dat3$factor <- factor(round(dat3$x/10,1)*10)  ##make bins from x
###########################################
##2.5.  Aggregate and tapply functions
###########################################
set.seed(111)
x <- rnorm(500)                  ##generate random numbers
y <- x + runif(500,-.3,.3)       ##related random numbers
dat3 <- data.frame(x=x,y=y)      ##create a data frame
dat3$factor <- factor(round(dat3$x/10,1)*10)  ##make bins from x
##aggregate y by the bins:
dat3.agg <-     aggregate(dat3$y,list(bin=dat3$factor),mean)
##aggregate x by the same bins:
dat3.agg$xvals <- aggregate(dat3$x,list(bin=dat3$factor),mean)$x
##use tapply to aggregate y by the bins:
dat3.tab <- tapply(dat3$y,list(bin=dat3$factor),mean)
dat3
###########################################
##2.5.  Aggregate and tapply functions
###########################################
set.seed(111)
x <- rnorm(500)                  ##generate random numbers
y <- x + runif(500,-.3,.3)       ##related random numbers
dat3 <- data.frame(x=x,y=y)      ##create a data frame
dat3$factor <- factor(round(dat3$x/10,1)*10)  ##make bins from x
##aggregate y by the bins:
dat3.agg <-     aggregate(dat3$y,list(bin=dat3$factor),mean)
##aggregate x by the same bins:
dat3.agg$xvals <- aggregate(dat3$x,list(bin=dat3$factor),mean)$x
##use tapply to aggregate y by the bins:
dat3.tab <- tapply(dat3$y,list(bin=dat3$factor),mean)
dat3.tab
dat3.agg
data(ChickWeight)
attributes(ChickWeight)$labels
head(ChickWeight)
length(table(ChickWeight$Chick))
plot(ChickWeight$Time,ChickWeight$weight,
ylab="Weight (g)",xlab="Age (days)")
cscheme <- c("red","darkgreen","black","orange")
plot(ChickWeight$Time,ChickWeight$weight,ylab="Weight (g)",xlab="Age (days)",
pch=16, col=cscheme[ChickWeight$Diet])
cw.agg <- tapply(ChickWeight$weight,
list(time=ChickWeight$Time,
diet=ChickWeight$Diet),mean)
cw.agg
times <- c(0,2,4,6,8,10,12,14,16,18,20,21)
times <- as.numeric(rownames(cw.agg))
plot(ChickWeight$Time,ChickWeight$weight,cex=.5,
ylab="Weight (g)",xlab="Age (days)",
pch=1, col=cscheme[ChickWeight$Diet])
matplot(times,cw.agg,add=T,type="l",lwd=3,
col=cscheme,lty=1)
cw.bysub<- tapply(ChickWeight$weight,
list(time=ChickWeight$Time,
chick=ChickWeight$Chick),mean)
cw.bysub[,1:10]
diets <- tapply(as.numeric(as.character(ChickWeight$Diet)),
list(chick=ChickWeight$Chick),median)
diets <- aggregate(as.numeric(as.character(ChickWeight$Diet)),
list(chick=ChickWeight$Chick),median)$x
diets
matplot(times,cw.bysub,col=cscheme[diets],type="l",lty=3,
ylab="Weight (mg)",xlab="Age (days)")
ChickWeight
matplot(times,cw.bysub,col=cscheme[diets],type="l",lty=3,
ylab="Weight (mg)",xlab="Age (days)")
segments(-10,0:15*25,25,0:15*25,lty=3,col="grey")
segments(-10,0:7*50,25,0:7*50,lty=1,col="grey")
points(ChickWeight$Time,ChickWeight$weight,cex=.5,
pch=1, col=cscheme[ChickWeight$Diet])
matplot(times,cw.agg,add=T,type="l",lwd=7, col="white",lty=1)
matplot(times,cw.agg,add=T,type="l",lwd=3, col=cscheme,lty=1)
title("Chick weights over time by diet")
legend(2,350,paste("Diet",c(1:4)),col=cscheme,lty=1,lwd=5,bg="white")
#Some ways to get basic information about a data structure
data(trees)  ##load the data
## look at some summary values
dim(trees)
str(trees)
attributes(trees)
summary(trees)
##try plotting to see what is going on
#pdf("c2f1.pdf",width=9,height=5)
par(mfrow=c(1,2))
matplot(trees)
boxplot(trees)
#dev.off()
data(trees)
View(trees)
dim(trees)
str(trees)
attributes(trees)
trees
trees[1:5,]
trees[10:15,]
10:15
c(14,15,10,12,13,11)
trees[c(14,15,10,12,13,11),]
rows <- c(14,15,10,12,13,11)
trees[rows,]
trees[10:15,]
trees[10:15,][5,6,1,3,2,4]
trees[10:15,][c(5,6,1,3,2,4),]
trees[10:15,][c(5,6,1,3,2,4),] -2
trees[10:15,-2][c(5,6,1,3,2,4),]
trees[10:15,][c(5,6,1,3,2,4),-2]
trees[10:15,][c(5,6,1,3,2,4),][,-2]
trees[10:15,][c(5,6,1,3,2,4),][,1:3]
trees[10:15,][c(5,6,1,3,2,4),][,c(1,3)]
trees[10:15,][c(5,6,1,3,2,4),][,c(3,1)]
trees
trees$Girth
trees[,1]
trees["Girth"]
order(trees["Girth"])
order(trees$Girth)
order(trees$Volume)
sort()
sort(trees$Volume)
trees[order(trees$Volume),]
rank(c(2,.5,1,1.5,2,2.5))
rank(trees$Girth)
rank(trees$Girth)/32
order(order(c(2,.5,1,1.5,2,2.5)))
party <-  c("R","R","D","R","R","D","D","D","R","R","D")
gender <- c("M","M","F","F","F","F","M","M","F","M","M")
vote   <- c("A","B","A","A","A","B","A","A","B","B","A")
survey <- data.frame(party,gender,vote)
survey
table(party,gender)
#Demo: who makes more in the political survey?
set.seed(111)
survey$income <- runif(11,20000,100000)
survey
table(party,survey$income)
table(party,round(survey$income,-4)
)
survey[order(survey$income)]
survey[order(survey$income),]
survey[order(survey$party),]
survey[order(survey$party),][1:6,]
mean(survey[order(survey$party),][1:6,]$income)
mean(survey[order(survey$party),][7:12,]$income)
mean(survey[order(survey$party),][7:11,]$income)
aggregate(survey$income,survey["party"],mean)
aggregate(survey$income,survey$party,mean)
aggregate(survey$income,list(survey$party),mean)
aggregate(survey$income,survey[,1:2],mean)
aggregate(survey$income,survey[,1:3],mean)
tapply(survey$income,survey[,1:2],mean)
aggegate(survey$income,survey[,1:2],mean)
aggregate(survey$income,survey[,1:2],mean)
matplot(tapply(survey$income,survey[,1:2],mean))
matplot(tapply(survey$income,survey[,1:2],mean),type="l")
matplot(tapply(survey$income,survey[,1:2],mean),type="o")
matplot(tapply(survey$income,survey[,1:2],median),type="o")
matplot(tapply(survey$income,survey[,1:2],sd),type="o")
matplot(tapply(survey$income,survey[,1:2],mean),type="o")
tapply(survey$income,survey[,1:3],mean)
aggregate(survey$income,survey[,1:3],mean)
survey
set.seed(111)
x <- rnorm(500)                  ##generate random numbers
y <- x + runif(500,-.3,.3)       ##related random numbers
dat3 <- data.frame(x=x,y=y)      ##create a data frame
dat3$factor <- factor(round(dat3$x/10,1)*10)  ##make bins from x
##aggregate y by the bins:
dat3.agg <-     aggregate(dat3$y,list(bin=dat3$factor),mean)
##aggregate x by the same bins:
dat3.agg$xvals <- aggregate(dat3$x,list(bin=dat3$factor),mean)$x
##use tapply to aggregate y by the bins:
dat3.tab <- tapply(dat3$y,list(bin=dat3$factor),mean)
dat3.agg
dat3.tab
set.seed(111)
dat3$factor2 <- floor(rank(dat3$x)/20)
dat3.agg2 <- aggregate(dat3$y,list(bin=dat3$factor2),mean)
dat3.agg2$xvals <-aggregate(dat3$x,list(bin=dat3$factor2),mean)$x
plot(dat3.agg$xvals,dat3.agg$x,col="blue",pch=16,
xlab="Mean values of x bins",ylab="Values of y")
col2 <- rgb(1,0,0,.6) #a see-through red
col1 <- rgb(0,0,1,.6) # a see-through blue
points(dat3.agg2$xvals,dat3.agg2$x,cex=3,col=col2,pch=17)
points(dat3.agg$xvals,dat3.agg$x,cex=1,col=col1,pch=16)
dat3
dat3
dat[1:5,]
dat3[1:5,]
plot(dat3$factor,dat3$x)
plot(as.numeri(dat3$factor),dat3$x)
plot(as.numeric(dat3$factor),dat3$x)
plot(as.numeric(dat3$factor),dat3$y)
dat3.agg2 <- aggregate(dat3$y,list(bin=dat3$factor2),mean)
dat3.agg2 <- aggregate(dat3$y,list(bin=dat3$factor2),mean)
dat3.agg2$xvals <-aggregate(dat3$x,list(bin=dat3$factor2),mean)$x
plot(dat3.agg$xvals,dat3.agg$x,col="blue",pch=16,
xlab="Mean values of x bins",ylab="Values of y")
col2 <- rgb(1,0,0,.6) #a see-through red
col1 <- rgb(0,0,1,.6) # a see-through blue
points(dat3$x,dat3$y)
dat3
head(dat3)
plot(x,y)
plot(dat3$factor,y)
plot(y,dat3$factor)
matplot(tapply(survey$income,survey[,1:2],mean),type="o")
plot(as.numeric(dat3$factor),y)
dat3$factor
as.numeric(dat3$factor)
as.numeric(as.character(dat3$factor))
as.character(dat3$factor)
as.numeric(as.character(dat3$factor))
as.integer(as.character(dat3$factor))
options <- c("control", "A-condition", "B-condition")
options
sample(options,20,replace=T)
factor(sample(options,20,replace=T))
factor(sample(options,20,replace=T),levels=options)
factor(sample(options,20,replace=T),levels=options)
a <- factor(sample(options,20,replace=T),levels=options)
a
as.numeric(a)
set.seed(100)
m <- matrix(runif(28),7,4)
m
##aggregate by row:
m
round(apply(m,1,var),3) ##By row
round(apply(m,2,var),3) ##By row
rowMeans(m)
round(apply(m,1,mean),3) ##By row
rowMeans(m)
par(mfrow=c(1,1))
matplot(t(m),type="p",pch=16,col="black")
points(mins,type="l")
data(ChickWeight)
attributes(ChickWeight)$labels
ChickWeight
ChickWeight[1:5,]
ChickWeight[50:60,]
ChickWeight$Chick
ChickWeight$Chick == 5
ChickWeight[ChickWeight$Chick == 5,]
ChickWeight[ChickWeight$Time == 10,]
ChickWeight$Time == 10
c("",".")[ChickWeight$Time == 10]
c("",".")[0+(ChickWeight$Time == 10)]
c("",".")[ChickWeight$Time == 10]
c("",".")[0+(ChickWeight$Time == 10)]
0+(ChickWeight$Time == 10)]
0+(ChickWeight$Time == 10)
1+(ChickWeight$Time == 10)
c("yes","no")[1+(ChickWeight$Time == 10)]
c("red","blue")[1+(ChickWeight$Time == 10)]
ChickWeight
head(ChickWeight)
plot(ChickWeight$weight)
plot(ChickWeight$weight,type="l")
plot(ChickWeight$Time,ChickWeight$weight,type="l")
plot(ChickWeight$Time,ChickWeight$weight)
plot(ChickWeight$Time,ChickWeight$weight,
ylab="Weight (g)",xlab="Age (days)")
cscheme <- c("red","darkgreen","black","orange")
cscheme
cscheme[ChickWeight$Diet]
ChickWeight$Diet
cscheme[1:4]
cscheme[c(1,1,1)]
cscheme[rep(1,10)]
cscheme[c(rep(1,10),rep(2,4)]
cscheme[c(rep(1,10),rep(2,4))]
ChickWeight$Diet
cscheme[ChickWeight$Diet]
ChickWeight$color <- cscheme[ChickWeight$Diet]
ChickWeight
plot(ChickWeight$Time,ChickWeight$weight,ylab="Weight (g)",xlab="Age (days)",
pch=16, col=cscheme[ChickWeight$Diet])
aggregate(ChickWeight$weight,ChickWeight[,c(2,4)],mean)
aggregate(ChickWeight$weight,list(ChickWeight$Day,ChickWeight$Diet),mean)
aggregate(ChickWeight$weight,list(ChickWeight$Time,ChickWeight$Diet),mean)
tapply(ChickWeight$weight,list(ChickWeight$Time,ChickWeight$Diet),mean)
matplot(tapply(ChickWeight$weight,list(ChickWeight$Time,ChickWeight$Diet),mean))
matplot(tapply(ChickWeight$weight,list(ChickWeight$Time,ChickWeight$Diet),mean), type="b")
matplot(tapply(ChickWeight$weight,list(ChickWeight$Time,ChickWeight$Diet),mean), type="l")
cw.agg <- tapply(ChickWeight$weight,
list(time=ChickWeight$Time,
diet=ChickWeight$Diet),mean)
cw.agg
plot(ChickWeight$Time,ChickWeight$weight,ylab="Weight (g)",xlab="Age (days)",
pch=16, col=cscheme[ChickWeight$Diet])
cw.agg
times <- c(0,2,4,6,8,10,12,14,16,18,20,21)
times
rownames(cw.agg)
as.numeric(rownames(cw.agg))
plot(ChickWeight$Time,ChickWeight$weight,cex=.5,
ylab="Weight (g)",xlab="Age (days)",
pch=1, col=cscheme[ChickWeight$Diet])
plot(ChickWeight$Time,ChickWeight$weight,cex=.5,
ylab="Weight (g)",xlab="Age (days)",
pch=2, col=cscheme[ChickWeight$Diet])
plot(ChickWeight$Time,ChickWeight$weight,cex=.5,
ylab="Weight (g)",xlab="Age (days)",
pch=1, col=cscheme[ChickWeight$Diet])
matplot(times,cw.agg,add=T,type="l",lwd=3,
col=cscheme,lty=1)
cw.bysub<- tapply(ChickWeight$weight,
list(time=ChickWeight$Time,
chick=ChickWeight$Chick),mean)
cw.bysub[,1:10]
diets <- tapply(as.numeric(as.character(ChickWeight$Diet)),
list(chick=ChickWeight$Chick),median)
diets <- aggregate(as.numeric(as.character(ChickWeight$Diet)),
list(chick=ChickWeight$Chick),median)$x
diets
matplot(times,cw.bysub,col=cscheme[diets],type="l",lty=3,
ylab="Weight (mg)",xlab="Age (days)")
segments(-10,0:15*25,25,0:15*25,lty=3,col="grey")
segments(-10,0:7*50,25,0:7*50,lty=1,col="grey")
points(ChickWeight$Time,ChickWeight$weight,cex=.5,
pch=1, col=cscheme[ChickWeight$Diet])
matplot(times,cw.agg,add=T,type="l",lwd=3, col=cscheme,lty=1)
matplot(times,cw.agg,add=T,type="l",lwd=7, col="white",lty=1)
matplot(times,cw.agg,add=T,type="l",lwd=3, col=cscheme,lty=1)
matplot(times,cw.agg,add=T,type="l",lwd=10, col="white",lty=1)
matplot(times,cw.agg,add=T,type="l",lwd=3, col=cscheme,lty=1)
matplot(times,cw.agg,add=T,type="l",lwd=10, col=rgb(0,0,0,.5),lty=1)
#erase back of line
matplot(times,cw.agg,add=T,type="l",lwd=10, col=rgb(1,1,1,.5),lty=1)
matplot(times,cw.agg,add=T,type="l",lwd=10, col=rgb(255,255,255,.5),lty=1)
matplot(times,cw.bysub,col=cscheme[diets],type="l",lty=3,
ylab="Weight (mg)",xlab="Age (days)")
##Lets add some gridlines
segments(-10,0:15*25,25,0:15*25,lty=3,col="grey")
segments(-10,0:7*50,25,0:7*50,lty=1,col="grey")
points(ChickWeight$Time,ChickWeight$weight,cex=.5,
pch=1, col=cscheme[ChickWeight$Diet])
#erase back of line
matplot(times,cw.agg,add=T,type="l",lwd=10, col=rgb(255,255,255,.5),lty=1)
matplot(times,cw.agg,add=T,type="l",lwd=3, col=cscheme,lty=1)
title("Chick weights over time by diet")
legend(2,350,paste("Diet",c(1:4)),col=cscheme,lty=1,lwd=5,bg="white")
se <- function(x){sd(x)/sqrt(length(x))}
se
se(runif(100))
