Repository navigation
Expand file tree
/
Copy pathMisc_practice.R
More file actions
91 lines (69 loc) · 2.31 KB
/
Copy pathMisc_practice.R
File metadata and controls
91 lines (69 loc) · 2.31 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
#R creating variables practice/ creating factors
yesno <- sample(c("yes", "no"), size=10, replace=T)
yesnofac <- factor(yesno)
relevel(yesnofac, ref="yes")
round(3.345, digits = 1)
#Reshaping data
install.packages("Hmisc")
library(reshape2)
library(Hmisc)
data(mtcars)
mtcars$carname <- rownames(mtcars)
head(mtcars$carname, 5)
#Melting Data
carMelt <- melt(mtcars, id=c("carname", "gear","cyl"),
measure.vars =c("mpg","hp"))
head(carMelt, 3) #shows mpg variable
tail(carMelt, 3) #shows hp variable
#Casting Data
cylData <- dcast(carMelt, cyl~variable, mean)
cylData #provides mean mpg & hp for each cyl
#Averaging values
head(InsectSprays)
names(InsectSprays)
tapply(InsectSprays$count, InsectSprays$spray, mean) #total counts by spray type
#Using split & lapply
spIns <- split(InsectSprays$count, InsectSprays$spray)
sprCount <- lapply(spIns, sum)
#Combining/Simplifying results:
simplify2array(sprCount)
#or
unlist(sprCount)
#or
sapply(spIns, sum)
#Plyr Package to combine data
library(plyr)
ddply(InsectSprays,.(spray), summarize, sum=sum(count))
#Dplyr practice
library(dplyr)
setwd("~/Desktop")
chicago <- readRDS("chicago.rds")
head(select(chicago, city:dptp))
head(select(chicago, -(city:dptp)))
#group_by years
new_chicago <- mutate(chicago, year=as.POSIXlt(date)$year + 1900)
years <- group_by(new_chicago, year)
summarize(years, pm25tmean2=mean(pm25tmean2, na.rm=T), o3=max(o3tmean2),
no2= median(no2tmean2))
#pipeline operator example
#grouping summaries by month
chicago %>% mutate(month=as.POSIXlt(date)$mon+1) %>%
group_by(month) %>% summarise(pm25tmean2 = mean(pm25tmean2),
o3 = max(o3tmean2), no2=median(no2tmean2))
#Merging dataframes with a common id
library(plyr)
df1 <- data.frame(id=sample(1:10), x=rnorm(10))
df2 <- data.frame(id=sample(1:10), y=rnorm(10))
arrange(join(df1,df2), id)
#merging multiple df w/ plyr
df3 <- data.frame(id=sample(1:10), z=rnorm(10))
dfList <- list(df1, df2, df3)
join_all(dfList)
######## Cleaning character vectors:
names(ability.cov) #Removing the period in "n.obs"
splitNames <- strsplit(names(ability.cov), "\\.")
splitNames[[3]][1]
firstElement <- function(x) {x[1]}
sapply(splitNames, firstElement)
#To keep the full name, but remove only one character:
sub("\\.", "", names(ability.cov)) #now is "nobs"