R 실습 학습 2017-10-26 레슨2 : 데이터 전처리

 

#### Creating the stat data frame ####

 

squad <- c(2,7,10,19,27)

date <- c("01/01/17", "01/16/17", "02/01/17", "02/12/17", "02/28/17")

gender <- c("M", "M", "M", "M", "F")

age <- c(25,31,24,24,97)

f1 <- c(4,3,2,3,2)

f2 <- c(4,3,5,1,2)

f3 <- c(5,2,5,3,1)

f4 <- c(4,4,5,NA,4)

f5 <- c(3,5,2,NA,2)

 

stat <- data.frame(squad, date, gender, age,

                   f1,f2,f3,f4,f5, stringsAsFactors = FALSE)

 

stat

stat <- data.frame(squad = c(2,7,10,19,27), 

                   date = c("01/01/17", "01/16/17", "02/01/17", "02/12/17", "02/28/17"),

                   gender = c("M", "M", "M", "M", "F"),

                   age = c(25,31,24,24,97),

                   f1 = c(4,3,2,3,2),

                   f2 = c(4,3,5,1,2),

                   f3 = c(5,2,5,3,1),

                   f4 = c(4,4,5,NA,4),

                   f5 = c(3,5,2,NA,2),

                   stringsAsFactors = FALSE)

 

stat

 

ls()

 

## the individual vectors are no longer needed

rm(squad, date, gender, age, f1,f2,f3,f4,f5)

 

ls()

 

 

#### Creating new variables ####

 

data <- data.frame(x1 = c(3,2,5,1),

                   x2 = c(3,4,6,2))

 

data

 

data$sumx <- data$x1 + data$x2

apply(data, 1, sum)

data$meanx <- (data$x1 + data$x2) / 2

apply(data, 1, mean)

data

 

data <- transform(data, sumx = x1 + x2, meanx = (x1 + x2) / 2)

 

data

 

ls()

rm()

 

# rm(list=ls())

 

 

 

# Recoding variables

stat

stat$age[stat$age == 97] <- NA

stat$age

 

# numeric -> categorical variables

stat$agecat[stat$age > 30] <- "Old"

stat$agecat[stat$age > 24 & stat$age <= 30] <- "Middle Aged"

stat$agecat[stat$age <= 24] <- "Young"

which(is.na(stat$agecat))

stat

 

# or use ifelse

stat$agecat <- with(stat, 

                    ifelse((stat$age > 30), "Old", 

                           ifelse((stat$age > 24 & stat$age <= 30), "Middle Aged", "Young")))

 

stat$agecat

stat

 

 

# or more compactly

 

stat <- within(stat, {

  agecat1 <- NA

  agecat1[age > 30] <- "Old"

  agecat1[age >= 25 & age <= 30] <- "Middle Aged"

  agecat1[age < 25] <- "Young"

})

 

stat

 

class(stat$agecat)

 

stat$agecat <- factor(stat$agecat, levels = c("Old", "Middle Aged", "Young"))

stat

table(stat$agecat)  # table drops NA by default

table(stat$agecat, useNA = "ifany")

?table

 

 

# Renaming variables with the reshape package

install.packages("reshape")

library(reshape)

 

colnames(stat)

 

stat <- rename(stat, c(squad = "squadID", date="testdate"))

 

stat

 

# or rename variables without using the reshape package

 

names(stat)

names(stat)[1] <- "squadID"

names(stat)[2] <- "testdate"

names(stat)[5:9] <- paste("field", 1:5, sqp="")

 

y <- c(1,2,3,NA)

is.na(y)

 

 

# Applying the is.na() function

is.na(stat[,5:9])

complete.cases(stat[,5:9])

 

 

# recode 97 to missing for the variable age

stat$age[stat$age == 97] <- NA

stat

 

# na.rm=TRUE option

x <- c(1,2,NA,3)

x

y <- sum(x); y

y <- sum(x, na.rm=TRUE); y

x

 

 

# Using na.omit() to delete incomplete observations

stat

newdata <- na.omit(stat)

newdata

경축! 아무것도 안하여 에스천사게임즈가 새로운 모습으로 재오픈 하였습니다.
어린이용이며, 설치가 필요없는 브라우저 게임입니다.
https://s1004games.com

complete.cases(stat)

stat[complete.cases(stat),]

 

# default format for inputting dates : yyyy-mm-dd

mydates <- as.Date(c("2016-01-27", "2020-02-13")); mydates

class(mydates)

 

# Converting character values to dates

 

strDates <- c("01/05/2000", "08/16/2000")

dates <- as.Date(strDates, "%m/%d/%Y"); dates

 

 

install.packages("chron")

library(chron)

Sys.Date()

date <- seq(Sys.Date(), by = 1, length.out = 7)

date

 

# Useful date functions

 

Sys.Date()

date()

 

# Use the format function to output dates in a special format,

# and to extract portions of dates

today <- Sys.Date()

today

format(today, format = "%B %d %Y")

today

format(today, format = "%A")

today

 

# Arithmetic operations on dates

 

startdate <- as.Date("2016-02-10")

enddate <- as.Date("2017-12-12")

days <- enddate - startdate; days

as.numeric(days)

difftime(enddate, startdate)

# difftime(Sys.Date()+ 1, as.Date("2016-05-07"))

 

# difftime function

# How old am I?

today <- Sys.Date()

dob <- as.Date("2010-10-10")

difftime(today,dob)

difftime(today, dob, units="weeks")

difftime(today, dob, units="days") # default

 

 

# converting dates to character variables

class(dates)

strDates <- as.character(dates)

strDates

class(strDates)

 

# Converting from one date type to another

 

a <- c(1,2,3)

a

is.numeric(a)

is.vector(a)

a <- as.character(a)

a

 

is.numeric(a)

is.vector(a)

is.character(a)

 

 

#### sort,order ####

# sorting a dataset 

 

# create a new dataset containing rows sorted from youngest to oldset squad

newdata <- stat[order(stat$age),]

newdata$age

 

# sort the rows into female follewed by male,

# and youngest to oldest within each gender.

attach(stat)

newdata <- stat[order(gender, -age), ]

newdata

detach()

 

 

# Selecting variables

 

newdata <- stat[, c(5:9)]; newdata

variables <- paste("field", 1:5, sep=""); variables

newdata <- stat[variables]; newdata

 

# Dropping variables

names(stat)[5:9] <- paste("field", 1:5, sep="")

names(stat)

variables <- names(stat) %in% c("field3", "field4"); variables

newdata <- stat[!variables]; newdata

 

# or use column numbers to drop variables

t(colnames(stat))

newdata <- stat[-c(8,9)]; newdata

 

 

# You could use the following to delete f3 and f4

# from the stat dataset (commented out so 

# the rest of the code in this file will work)

#

# stat$f3 <- stat$f4 <- NULL

 

newdata <- stat[1:3,]; newdata

 

newdata <- stat[which(stat$gender == "M" &

                        stat$age > 30),]

 

newdata

 

attach(stat)

newdata <- stat[which(gender == "M" & age > 30),]

detach(stat)

 

 

newdata <- subset(stat, gender=="M" & age > 30); newdata

 

# Selecting observations based on dates

 

stat$date <- as.Date(stat$date, "%m/%d/%Y");

startdate <- as.Date("2017-02-01")

enddate <- as.Date("2017-02-28")

newdata <- stat[stat$date >= startdate & 

                  stat$date <= enddate,]

 

newdata

 

# Using the subset() function

 

newdata <- subset(stat, age >= 35 | age < 24, 

                  select = c(field1, field2, field3, field4 ))

 

newdata <- subset(stat, gender=="M" & age > 25,

                  select = gender:field4)

 

newdata

 

 

 

본 웹사이트는 광고를 포함하고 있습니다.
광고 클릭에서 발생하는 수익금은 모두 웹사이트 서버의 유지 및 관리, 그리고 기술 콘텐츠 향상을 위해 쓰여집니다.
번호 제목 글쓴이 날짜 조회 수
공지 오라클 기본 샘플 데이터베이스 졸리운_곰 2014.01.02 86140
공지 [SQL컨셉] 서적 "SQL컨셉"의 샘플 데이타 베이스 SAMPLE DATABASE of ORACLE 가을의 곰을... 2013.02.10 78637
공지 [G_SQL] Sample Database 가을의 곰을... 2012.05.20 95366
42 블록체인 기반 플랫폼 비즈니스를 이해하자 졸리운_곰 2017.09.10 1553
41 다운타임 없는 서비스 구현 패턴 file 졸리운_곰 2017.09.10 1229
40 테이블의 수직분할과 수평분할에 대한 이해 file 졸리운_곰 2017.09.10 3908
39 정규화와 응집도에 대한 고찰 file 졸리운_곰 2017.05.28 1782
38 머신러닝 새 도전…“클라우드를 벗어나라" file 졸리운_곰 2017.05.28 1711
37 보안성 높이는 공공 거래장부 블록체인 file 졸리운_곰 2017.05.28 1757
36 디지털, 속도의 전쟁 VS 데이터, 품질의 전쟁 file 졸리운_곰 2017.05.05 1515
35 04. 데이터 모델링의 3단계 진행 file 졸리운_곰 2016.03.15 1659
34 데이터베이스 설계의 기본 원리.pdf file 졸리운_곰 2016.03.15 2353
33 실체유형(Entity Type) 정의 사항 및 도출 file 졸리운_곰 2015.05.21 2197
32 마농의 SQL 백문백답: 단순하고 쉽게 작성하는 SQL 노하우 [1회] file 졸리운_곰 2015.05.21 1829
31 sql개발자-sql전문가자격시험 시험 예제.pdf file 졸리운_곰 2015.02.15 2302
30 데이터베이스 선정에는 비밀이 있다 - 4부 졸리운_곰 2015.01.15 2095
29 데이터베이스 선정에는 비밀이 있다 - 3부 졸리운_곰 2015.01.15 2042
28 데이터베이스 선정에는 비밀이 있다 - 2부 졸리운_곰 2015.01.15 1827
27 데이터베이스 선정에는 비밀이 있다 - 1부 졸리운_곰 2015.01.15 2391
26 지금 우리에게 필요한 것은 데이터베이스 성능 최적화이다 (2부) 졸리운_곰 2015.01.15 1484
25 우리에게 필요한 것은 데이터베이스 성능 (1부) 졸리운_곰 2015.01.15 1773
24 21회 결과 secret 졸리운_곰 2014.11.10 0
23 [데이터아키텍쳐준전문가] 시험 fail 자료 secret 졸리운_곰 2014.08.31 0
대표 김성준 주소 : 경기 용인 분당수지 U타워 등록번호 : 142-07-27414
통신판매업 신고 : 제2012-용인수지-0185호 출판업 신고 : 수지구청 제 123호 개인정보보호최고책임자 : 김성준 sjkim70@stechstar.com
대표전화 : 010-4589-2193 [fax] 02-6280-1294 COPYRIGHT(C) stechstar.com ALL RIGHTS RESERVED