Differences
This shows you the differences between two versions of the page.
| Both sides previous revision Previous revision Next revision | Previous revision | ||
|
r_programming_gl [2014/10/30 18:10] glaroc |
r_programming_gl [2014/11/21 21:31] (current) glaroc old revision restored (2014/11/21 15:54) |
||
|---|---|---|---|
| Line 1: | Line 1: | ||
| + | ====== Knitr ====== | ||
| + | Knitr is a package that can be used to generate dynamic reports or web pages from R code. The code is evaluated at the moment the report is generated. | ||
| + | |||
| + | Code can be easily written in RStudio use the Markdown language. View this page in Markdown language, and view the resulting web page. | ||
| + | |||
| + | |||
| ====== Data Table ====== | ====== Data Table ====== | ||
| Data table is a very useful package in R which allows to facilitate and to improve the efficiency of certain operations in R. Data tables are just like data frames. You can even create them from data frames. | Data table is a very useful package in R which allows to facilitate and to improve the efficiency of certain operations in R. Data tables are just like data frames. You can even create them from data frames. | ||
| Line 9: | Line 15: | ||
| </code> | </code> | ||
| - | <code rsplus> | + | Generate very long data frame with one column with letters, and one column with random numbers |
| - | library(reshape2) | + | <file rsplus> |
| - | library(data.table) | + | mydf=data.frame(a=rep(LETTERS,each=1e5),b=rnorm(26*1e5)) |
| - | library(plyr) | + | </file> |
| - | mydf=data.frame(a=rep(LETTERS,each=1e6),b=rnorm(26*1e6)) | + | |
| + | Convert the data frame to a data table format. | ||
| + | <file rsplus> | ||
| mydt=data.table(mydf) | mydt=data.table(mydf) | ||
| + | </file> | ||
| + | |||
| + | Each data table has to be assigned a key, which is one (or more) of the columns from the table. This key defines the basis for the organization and the sorting of the table. | ||
| + | <file rsplus> | ||
| setkey(mydt,a) | setkey(mydt,a) | ||
| + | </file> | ||
| + | |||
| + | Once the key is set, we can return all rows with column a (the key) equal to F | ||
| + | <file rsplus> | ||
| mydt['F'] | mydt['F'] | ||
| - | # Returns all rows with column a (the key) equal to F | + | </file> |
| + | |||
| + | Gives the mean value of column b for each letter in column a. | ||
| + | <file rsplus> | ||
| mydt[,mean(b),by=a] | mydt[,mean(b),by=a] | ||
| - | # Gives the mean value of column b for each letter in column a. | + | </file> |
| - | # Compare | + | |
| + | Let's compare the performance of Data table with other methods to achieve the same thing. | ||
| + | <file rsplus> | ||
| system.time(t1<-mydt[,mean(b),by=a]) | system.time(t1<-mydt[,mean(b),by=a]) | ||
| - | # 0.314 secs | + | </file> |
| - | # With tapply() | + | |
| + | **With tapply()** | ||
| + | <file rsplus> | ||
| system.time(t2<-tapply(mydf$b,mydf$a,mean)) | system.time(t2<-tapply(mydf$b,mydf$a,mean)) | ||
| - | # 7.239 secs | + | </file> |
| - | # With reshape2 | + | |
| + | **With reshape2** | ||
| + | <file rsplus> | ||
| + | library(reshape2) | ||
| meltdf=melt(mydf) | meltdf=melt(mydf) | ||
| system.time(t3<-dcast(meltdf,a~variable,mean)) | system.time(t3<-dcast(meltdf,a~variable,mean)) | ||
| - | # 4.453 secs | + | </file> |
| - | # With reshape plyr | + | |
| + | **With plyr** | ||
| + | <file rsplus> | ||
| + | library(plyr) | ||
| system.time(t4<-ddply(mydf,.(a),summarize,mean(b))) | system.time(t4<-ddply(mydf,.(a),summarize,mean(b))) | ||
| - | # 2.288 secs | + | </file> |
| - | # With a FOR loop | + | |
| + | **With sqldf**. This package allows one to write Structured Query Language commands to perfom queries on a data frame. | ||
| + | <file rsplus> | ||
| + | library(sqldf) | ||
| + | system.time(t5<-sqldf('SELECT a, avg(b) FROM mydf GROUP BY a')) | ||
| + | </file> | ||
| + | |||
| + | **With a basic FOR loop** | ||
| + | <file rsplus> | ||
| ti1<-proc.time() | ti1<-proc.time() | ||
| - | t5<-data.frame(letter=unique(mydf$a),mean=rep(0,26)) | + | t6<-data.frame(letter=unique(mydf$a),mean=rep(0,26)) |
| - | for (i in t5$letter ){ | + | for (i in t6$letter ){ |
| - | t5[t5$letter==i,2]=mean(mydf[mydf$b==i,2]) | + | t6[t6$letter==i,2]=mean(mydf[mydf$a==i,2]) |
| } | } | ||
| eltime<-proc.time()-ti1 | eltime<-proc.time()-ti1 | ||
| - | </code> | + | eltime |
| + | </file> | ||
| + | |||
| + | **With a parallelized FOR loop** | ||
| + | <file rsplus> | ||
| + | library(foreach) | ||
| + | library(doMC) | ||
| + | registerDoMC(4) #Four-core processor | ||
| + | ti1<-proc.time() | ||
| + | t7<-data.frame(letter=unique(mydf$a),mean=rep(0,26)) | ||
| + | t7[,2] <- foreach(i=t7$letter, .combine='c') %dopar% { | ||
| + | mean(mydf[mydf$a==i,2]) | ||
| + | } | ||
| + | eltime<-proc.time()-ti1 | ||
| + | eltime | ||
| + | </file> | ||
| + | |||
| + | ====== RgoogleMaps! ====== | ||
| + | |||
| + | <file rsplus> | ||
| + | library(RgoogleMaps) | ||
| + | myhome=getGeoCode('Olympic stadium, Montreal'); | ||
| + | mymap<-GetMap(center=myhome, zoom=14) | ||
| + | PlotOnStaticMap(mymap,lat=myhome['lat'],lon=myhome['lon'],cex=5,pch=10,lwd=3,col=c('red')); | ||
| + | </file> | ||
| + | |||
| + | ====== Taxize ====== | ||
| + | <file rsplus> | ||
| + | library(taxize) | ||
| + | spp<-tax_name(query=c("american beaver"),get="species") | ||
| + | fam<-tax_name(query=c("american beaver"),get="family") | ||
| + | correctname <- tnrs(c("fraxinus americanus")) | ||
| + | cla<-classification("acer rubrum", db = 'itis') | ||
| + | </file> | ||
| + | |||
| + | ====== Spocc ====== | ||
| + | <file rsplus> | ||
| + | library(spocc) | ||
| + | occ_data <- occ(query = 'Acer nigrum', from = 'gbif') | ||
| + | mapggplot(occ_data) | ||
| + | </file> | ||
| + | |||
| + | Combine spocc and RgoogleMaps | ||
| + | <file rsplus> | ||
| + | occ_data <- occ(query = 'Puma concolor', from = 'gbif') | ||
| + | occ_data_df=occ2df(occ_data) | ||
| + | occ_data_df<-subset(occ_data_df,!is.na(latitude) & latitude!=0) | ||
| + | mymap<-GetMap(center=c(mean(occ_data_df$latitude),mean(occ_data_df$longitude)), zoom=2) | ||
| + | PlotOnStaticMap(mymap,lat=occ_data_df$latitude,lon=occ_data_df$longitude,cex=1,pch=16,lwd=3,col=c('red')); | ||
| + | </file> | ||
| + | |||
| + | ====== geonames ====== | ||
| + | |||
| + | <file rsplus> | ||
| + | library(geonames) | ||
| + | options(geonamesUsername="glaroc") | ||
| + | res<-GNsearch(q="Mont Saint-Hilaire") | ||
| + | res[,c('toponymName','fclName')] | ||
| + | dc<-GNcities(45.4, -73.55, 45.7, -73.6, lang = "en", maxRows = 10) | ||
| + | dc[,c('toponymName')] | ||
| + | </file> | ||
| + | |||
