## ----setup, include=FALSE--------------------------------------------------------------------------------------------------------------------------------------------------------------- knitr::opts_chunk$set(echo = TRUE) ## --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- library("dslabs") data(murders) head(murders) new_names<-ifelse(nchar(murders$stat)>8, murders$abb, murders$state) new_names ## --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- # Write a function compute_s_n with argument n that for any given n computes the sum of 1 + 2^2 + ...+ n^2 compute_s_n <- function(n){ x <- 1:n sum(x^2) } # Report the value of the sum when n=10 compute_s_n(10) ## ----Simulation------------------------------------------------------------------------------------------------------------------------------------------------------------------------- ##simulate 30 samples from N(0,1) set.seed(313) sam1=rnorm(30) ##equally split the sample, one as the treatment group and the other ##as the control group group<-rep(c("grp1", "grp2"), each=15) #### Shuffle the group label sgroup<-sample(group) group1=sam1[which(sgroup=="grp1")] group2=sam1[which(sgroup=="grp2")] ## ----pvalue----------------------------------------------------------------------------------------------------------------------------------------------------------------------------- ##perform t.test for the two samples t.test(group1,group2) ## ----uniform, fig.width=4, fig.height=4------------------------------------------------------------------------------------------------------------------------------------------------- ##generate 1000 samples from uniform distribution hist(runif(1000),freq=F,main="Histogram of 1000 random samples from uniform", xlab="Samples") ## ----permutation test, fig.width=4, fig.height=4, fig.align="center"-------------------------------------------------------------------------------------------------------------------- ##p-value of 1000 t-tests getpvalue=function(iteration){ g1=sample(1:30,15);g2=(1:30)[-g1] group1=sam1[g1]; group2=sam1[g2] ##perform t.test for the two samples return(t.test(group1,group2)$p.value) } simpvalue<-sapply(1:1000, getpvalue) ## ----plotp, fig.width=4, fig.height=4, fig.align="center"------------------------------------------------------------------------------------------------------------------------------- ##histogram hist(simpvalue,xlim=c(0,1),freq = F, main=" Distribution of p values under the null",xlab="p value") ## --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- ##(a) string<-sample(c('A','C','G','T'),10000,replace=TRUE) ##(b) table(string) ##(c) ktuple<-function(x, k){ ans<-NULL for (i in 1:(length(x)-k+1)){ tmp<-paste(x[i:(i+k-1)], collapse="") ans<-c(ans, tmp) } tab<-table(ans) return(tab) } ktuple(string, k=3) ## --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- myfactorial<-function(x){ if (x <0) stop("x needs to be an integer > 0") if (x==0){ ans<-1 } else{ ans<-1 for (i in 1:x){ ans<-ans*i } } return(ans) } myfactorial(10) ## --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- myfac2<-function(x){ if (x <0) stop("x needs to be an integer > 0") if (x==0){ ans<-1 } else{ txt<-paste(1:x, collapse="*") print(paste(txt, "=", sep="")) ans<-eval(parse(text=txt)) } return(ans) } myfac2(10)