R 실습 학습 2017-10-26 레슨2 :

데이터 입출력 및 기초통계

 

# condition statement #

 

#### Grouped expression ####
{
  a <- 2
  b <- 1:9
  a * b
}


#### Writing function ####
# ploynomial function
f <- function(x) x^2 + 1
f
x <- seq(-1,1, 0.1)
y <- f(x)
plot(x,y, type="l")
?plot


# odd-even function
odd.even <- function(x) {
  if(x %% 2 == 1) y <- "odd" else y <- "even"
  print(y)
}

odd.even(12)
odd.even

 

#### Control structures ####
# If condition
x <- 75
if (x >= 90) "A"
if (x >= 90) "A" else "B"
if (x >= 90) "A" else if (x >= 80) "B" else "C"

# example : median
x <- c(3,6,4,7,5,6,11,4,7,9)
length(x)
x.srt <- sort(x)
x.srt
x.len <- length(x)
if(x.len %% 2 == 1) {
  x.mode <- x.srt[(x.len + 1) / 2]
} else {x.mode <- ((x.srt[x.len / 2] * x.srt[x.len /2 + 1])/2)
}

x.mode
(x.len + 1) / 2
x.srt[(x.len + 1)/2]
x.len / 2
m <- function(x) {
  if(x.len %% 2 == 1) {
    x.mode <- x.srt[(x.len + 1) / 2]
  } else { x.mode <- ((x.srt[x.len / 2] + x.srt[x.len / 2 + 1]) / 2)
  }
  x.mode
}

m(x)

median(x)

#### looping ####
# for
for( i in c(1,3,4)) {
  print(x)
}

 

for (i in 10^(0:4)) print (sum(1:i))

# or

for (i in 10^(0:4)) {
  print(sum(1:i))
}

10^(0:4)

for(i in 1:3) {
  for(j in 4:6) {
    print( i * j )
  }
}


x <- matrix(NA, 3,3)
x
for (i in 1:3) {
  for (j in 1:3) {
    x[i,j] <- i * j
  }
}
x

x <- colors()  # colors() : character vector of length 657
for(i in 1:length(x)) {
  if(i %% 100 == 0) {
    cat("x[",i,"]:", x[i], "\n", sep = "")  # "\n" for new line
  }
}
x


# while
# repeats only when condition is TRUE, stops repeating when F
a <- 5
while(a > 0) {
  print(a)
  a <- a - 1
}
a


i <- 1
while (i < 6) {
  print(i*2)
  i <- i + 1
}


# compare the above while loop with for loop
for(i in 1:5) {
  print (i * 2)
}

x <- colors()
i <- 1
while ( i <= length(x) ) {
  if( x[i] == "orange" ) {
    cat("x[", i, "]:", x[i], "\n", sep="")
    break
  }
  i <- i + 1
}

x


## DO NOT RUN THIS INFINITE LOOPS CODE ##
## IF YOU RUN THIS CODE, CLICK RED BUTTON ON THE UPPER-RIGHT SIDE OF THE CONSOLE ##
i <- i
while(T) {
  print (i * 2)
  i <- i + 1
}

## OK TO RUN CODE FROM HERE ##
i <- i
while(T) {
  print (i * 2)
  i <- i + 1
  if(i > 5) break  # use if and break to avoid infinite loops
}


#### cat function ####
x <- 11:20; x
y <- 11:15
x;y

cat(x, y, sep = "-")
print(x, y)
cat("x=", x, sep = "")

# cat function can be used to write data
getwd()
setwd("/home/comphy/RStudioStudy/tmp")  # set some dir
getwd()
list.files()
cat("x=", x, "\n", sep="", file="cat.txt")
cat("x=", y, "\n", sep="", file="cat.txt")
cat("y=", y, "\n", sep="", file="cat.txt", append=T)
cat("y=", 1:15, sep="", file="cat.txt", append=T)
cat("y=", 1892, "\n", sep="", file="cat.txt", append=T)


#### Reading data ####

# Accessing built-in datasets
data()
data(women)  # same as data("women")
# or
data(women, package="datasets")
head(women)
# Reading data from files - Example

# Comma-separated values; CSV
airquality_csv <- read.table("airquality.csv", header=T, sep=",")
# or
airquality_csv <- read.csv("airquality.csv")
airquality_csv

# CSV2
airquality_csv2 <- read.table("airquality.csv2", header=T, sep=";", dec=",") # ????????
# or
airquality_csv2 <- read.csv2("airquality.csv2")
airquality_csv2
?read.table

# Delimited file - defaulting to the TAB character for the delimiter.
airquality_txt <- read.table("airquality.txt", header=T, sep="\t")
airquality_txt <- read.table("airqualiry.txt", header=T, sept="\t", na.strings=".") # if the missing variables
# or
airquality_txt <- read.delim("airquality.txt")
airquality_txt

#### Writing Data ####

경축! 아무것도 안하여 에스천사게임즈가 새로운 모습으로 재오픈 하였습니다.
어린이용이며, 설치가 필요없는 브라우저 게임입니다.
https://s1004games.com

# example : TAB Delimited file & CSV
data(iris)
head(iris)
out.data <- iris[1:4]
# TAB character for the delimiter
# row.names = T by default, which prints row label.
write.table(out.data, "iris.txt", quote=F, sep="\t",
            row.names=F, col.names=T)
# CSV
write.table(out.data, "/home/comphy/RStudioStudy/tmp/test123.txt", quote = T, #sep=","
            row.names=F, col.names = T)

?write.table

 

# R cookbook charpter 12.

#inserting data into a Vector
# append(vector, newvalues, after=n)
x <- append(1:10, NA, after=3) # the new items will be inserted at the position given by after.
x
is.na(x)
which(is.na(x))
append(1:10, 0, after = 0) # after = 0 means insert the new itmes at the head of the vector.

# Combining Multiple Vectors into One Vector and a Factor.
comb <- stack(list(group1 = 1:2, group2 = 3:4))
list(group1 = 1:2, group2 = 3:4)
print(comb)
str(comb)

# comb <- stack(list(v1=v1, v2 = v2, v3 = v3)) #Combine 3 Vectors

#### Appending Rows to a Data Frame ####
# rbind(vectors or matrices)


# Appending Rows to a Data Frame
dt.ori <- data.frame(x = 1:3, y = letters[1:3])
dt.tmp <- data.frame(x = 4:5, y = letters[4:5])
dt <- rbind(dt.ori, dt.tmp)
dt

# Selecting Rows and Columns More Easily
# subset(x, subset, select)
data("airquality")
head(airquality)
subset(airquality, Temp < 80, select = c(Ozone, Temp))
subset(airquality, Day == 1, select = - Temp)
subset(airquality, select = 1:3)
subset(airquality, select = Ozone:Wind)


# Removing NAs from a Data Frame
df <- data.frame(x = c(NA, 2, 3), y = c(0,10, NA)); df
cumsum(df)  # cumsum fails if the input contains NA values
df.clean <- na.omit(df)
df.clean
mean(df$x)
mean(df$x, na.rm = T)


# Combining Two Data Frames
dt.ori <- data.frame(x = 1:3, y = letters[1:3]); dt.ori
dt <- cbind(dt.ori, z = "a")
dt
dt <- cbind(dt.ori, i = 1:4) # error
dt <- cbind(dt.ori, i = 1:3 ) # match value
dt.ori
rbind(dt.ori, z = 0)     # Check out for the recycling rule
rbind(dt.ori, data.frame(x = 0, y = letters[4])) # add x = 0, y = 'd'

# Accessing Data Frame Contents More Easily
search()   # Environment of current R workplace
data(women)
head(women)
# attach(data.frame or list )
# dettach()

summary(women$height)

head(women)
attach(women) # for repetitive access, use attach to insert data frame into your search list.
search()  # now can refer to the data frame column by name without mentioning the data frame.
summary(height)   # The same variable now available by name

detach()   # remove the second location in the search list
search()
summary(height)


#### factor : category : group ####
# groups <- split(x,f)
library(MASS)
?MASS
str(Cars93)  # Origin has two levels : USA and non-USA
?Cars93
group <- split(Cars93$MPG.city, Cars93$Origin) # split the MPG data according to origin
group
summary(group)
class(group)
median(group[[1]])
median(group[[2]])
class(group[1]) # list
class(group[[1]]) # integer
group$USA[1]
group$USA
group[[1]][1]
group[1]
group
class(group)
group[1]
group[2]
group$USA[1]

 

#### function : apply row or column ####
## apply : function
## result <- apply(martirx or array, I, function)
data(co2)
?co2
str(co2)
plot(co2)
?apply
means <- apply(matrix(co2, ncol = 12, byrow = T), 1, mean)  # {1:row}, {2:column}
dim(matrix(co2, ncol = 12, byrow = T))
means # 39 years
names(means) <- 1959:1997
length(1959:1997)
plot(means)


#### Applying a Function to Every Column
data(co2)
means <- apply(matrix(co2, ncol = 12, byrow = T), 2, mean)
names(means) <- 1:12
plot(means)

# For a data frame
data(iris)
apply(iris[1:4], 2, mean)
sapply(iris[1:4], mean)  # sapply returns the results in a vector if possible.
lapply(iris[1:4], mean)  # lapply always returns the reuslt in list.
apply(iris[1:4], 2, summary)
sapply(iris[1:4], summary)
lapply(iris[1:4], summary)


## Applying a Function to Groups of Data
# tapply function
# tapply(vector, list of one or more factors, function, ...)
## contingency table from data.frame : array with named dimnames
data("warpbreaks")
?warpbreaks
str(warpbreaks)
head(warpbreaks)
?tapply
tapply(warpbreaks$breaks, warpbreaks[-1], mean)


## Applying a Function to Group of Rows
head(iris)
by(iris, iris[5], summary)  ## splite data according to Species and call summary for the three groups


## Applying a Function to Parallel Vectors or Lists
mapply(rep, 1:4, 4:1)
rep(1,4)
rep(2:3)
mapply(rep, times=1:4, x = 4:1)
mapply(seq, from = 1, to = 1:10)
?mapply

## Getting the Length of a String
nchar("Gerrard") # nchar returns a length of a string
msn <- c("Messi", "Suarez", "Neymar")
nchar(msn)

length("Gerrard")
length(msn)  # length returns a length of a vector

# Concatenating Strings
class(paste(1:12))
paste(1:12)
as.character(1:12)
paste("A", 1:6, sep="")
msn <- c("Messi", "Suarez", "Neymar")
shirtnumber <- c(10, 9, 11)
paste(msn, shirtnumber, sep="-")
paste(msn, "loves", "Barcelona.")
paste(msn, "loves", "Barcelona", collapse=", and ") # collapse parameter defines top-level
paste("Today is", Sys.Date())
paste("Today is", date())
Sys.Date() - 2


# Extracting Substrings
substr("Statistics", 1, 4)
substr("Statistics", 7,10)

bpl <- c("Arsenal", "Tottenham", "Chelsea")
substr(bpl, 1,3)  # extract first 3 characters of each string

cities <- c("New York, NY", "Los Angeles, CA", "Peoria, IL")
substr(cities, nchar(cities) - 1, nchar(cities))
nchar(cities)


# Replacing Substrings
manager <- c("Mourinho is the normal one. Mourinho is the special one.")
sub("Mourinho", "Klopp", manager) # sub replaces the first instance of a string
gsub("one", "manager", manager)

# Creating a Sequence of Dates
s <- as.Date("2016-01-01"); e <- as.Date("2016-01-01") # as.Date("yyyy-mm-dd") format
seq(from = s, to = e, by = 1) # one month of dates
seq.Date(from=s, to=e, by = 1)
# {from: starting date}, {by = increment}, {length.out : number of dates}
seq(from=s, by = 1, length.out = 7) # Dates, one week apart
seq(from=s, by = "month", length.out = 6) # First of the month for 6 months
seq(from=s, by = "3 months", length.out = 4) # Quartery dates for one year
seq(from=s, by = "year", length.out = 10) # Year-start dates for one decade

# Peeking at Your Data
head(women)
head(women, 10)
tail(women)

# Finding the Position of a Particular Value
match("s", letters)
letters
length(letters)
class(letters)
letters[19]

# Sorting a Data Frame
library(MASS)
data("Cars93")
names(Cars93)
t(names(Cars93))
Cars93.sub <- subset(Cars93, select=c(1,2,5))
attach(Cars93.sub)
head(Cars93.sub)
summary(Cars93.sub$Price)
Price
length(Price)
order(Price)
order(Price, decreasing = T)
rev(order(Price))
sort(Price)
sort(Price, decreasing = T)
rev(sort(Price))
Cars93.srt <- Ca
rs93.sub[order(Price), ]
head(Cars93.srt); tail(Cars93.srt)


 

 

 

본 웹사이트는 광고를 포함하고 있습니다.
광고 클릭에서 발생하는 수익금은 모두 웹사이트 서버의 유지 및 관리, 그리고 기술 콘텐츠 향상을 위해 쓰여집니다.
번호 제목 글쓴이 날짜 조회 수
공지 오라클 기본 샘플 데이터베이스 졸리운_곰 2014.01.02 86736
공지 [SQL컨셉] 서적 "SQL컨셉"의 샘플 데이타 베이스 SAMPLE DATABASE of ORACLE 가을의 곰을... 2013.02.10 79062
공지 [G_SQL] Sample Database 가을의 곰을... 2012.05.20 95823
42 MySQL 데이터베이스 기초 file 졸리운_곰 2018.07.05 1921
41 MySQL 기본 사용법 및 예제 졸리운_곰 2018.07.05 1207
40 COUNT() and GROUP BY 졸리운_곰 2018.07.05 1153
39 한 행에 중복된 값을 겹치지않게 count 해오는법(distinct , group by) 졸리운_곰 2018.07.05 1068
38 MYSQL GROUP BY 후 ROW COUNT file 졸리운_곰 2018.07.05 1175
37 group by로 해서 묶은 그룹의 count의 총 수(총 row수) 뽑기 file 졸리운_곰 2018.07.05 773
36 MySQL - 일별통계, 주간통계, 월간통계 졸리운_곰 2018.07.05 3020
35 mysql select 한 값을 insert 하는 sql 졸리운_곰 2018.07.02 1309
34 Auditing your MySQL Data 졸리운_곰 2018.07.02 1008
33 How To: Use MySQL triggers to log table changes 졸리운_곰 2018.07.02 1100
32 MySQL - History Tables 이력관리 / 히스토리 테이블 졸리운_곰 2018.07.02 2056
31 MariaDB 10의 NoSQL 기능과 MySQL의 Json 관련 UDF 졸리운_곰 2018.06.22 1308
30 [MySQL] Select 결과 Update하는 SQL 작성 file 졸리운_곰 2018.06.20 2472
29 조건에 맞게 select 한 후 update 시키기 졸리운_곰 2018.06.20 1151
28 MySQL (select) UPDATE file 졸리운_곰 2018.06.20 1065
27 MySQL에서 중복 값 찾기 졸리운_곰 2018.06.15 957
26 mysql case문 사용하기 졸리운_곰 2018.06.14 927
25 [MySQL] UPDATE 시 에러코드 1175 처리 file 졸리운_곰 2018.05.30 821
24 MySQL 데이터형 및 크기 졸리운_곰 2018.05.13 1130
23 MySQL OR MariaDB에서 프로시저(Procedure)를 만들어보자. 졸리운_곰 2018.03.25 1166
대표 김성준 주소 : 경기 용인 분당수지 U타워 등록번호 : 142-07-27414
통신판매업 신고 : 제2012-용인수지-0185호 출판업 신고 : 수지구청 제 123호 개인정보보호최고책임자 : 김성준 sjkim70@stechstar.com
대표전화 : 010-4589-2193 [fax] 02-6280-1294 COPYRIGHT(C) stechstar.com ALL RIGHTS RESERVED