R 실습 학습 2017-10-26 레슨2 : 데이터 전처리

 

#### Creating the stat data frame ####

 

squad <- c(2,7,10,19,27)

date <- c("01/01/17", "01/16/17", "02/01/17", "02/12/17", "02/28/17")

gender <- c("M", "M", "M", "M", "F")

age <- c(25,31,24,24,97)

f1 <- c(4,3,2,3,2)

f2 <- c(4,3,5,1,2)

f3 <- c(5,2,5,3,1)

f4 <- c(4,4,5,NA,4)

f5 <- c(3,5,2,NA,2)

 

stat <- data.frame(squad, date, gender, age,

                   f1,f2,f3,f4,f5, stringsAsFactors = FALSE)

 

stat

stat <- data.frame(squad = c(2,7,10,19,27), 

                   date = c("01/01/17", "01/16/17", "02/01/17", "02/12/17", "02/28/17"),

                   gender = c("M", "M", "M", "M", "F"),

                   age = c(25,31,24,24,97),

                   f1 = c(4,3,2,3,2),

                   f2 = c(4,3,5,1,2),

                   f3 = c(5,2,5,3,1),

                   f4 = c(4,4,5,NA,4),

                   f5 = c(3,5,2,NA,2),

                   stringsAsFactors = FALSE)

 

stat

 

ls()

 

## the individual vectors are no longer needed

rm(squad, date, gender, age, f1,f2,f3,f4,f5)

 

ls()

 

 

#### Creating new variables ####

 

data <- data.frame(x1 = c(3,2,5,1),

                   x2 = c(3,4,6,2))

 

data

 

data$sumx <- data$x1 + data$x2

apply(data, 1, sum)

data$meanx <- (data$x1 + data$x2) / 2

apply(data, 1, mean)

data

 

data <- transform(data, sumx = x1 + x2, meanx = (x1 + x2) / 2)

 

data

 

ls()

rm()

 

# rm(list=ls())

 

 

 

# Recoding variables

stat

stat$age[stat$age == 97] <- NA

stat$age

 

# numeric -> categorical variables

stat$agecat[stat$age > 30] <- "Old"

stat$agecat[stat$age > 24 & stat$age <= 30] <- "Middle Aged"

stat$agecat[stat$age <= 24] <- "Young"

which(is.na(stat$agecat))

stat

 

# or use ifelse

stat$agecat <- with(stat, 

                    ifelse((stat$age > 30), "Old", 

                           ifelse((stat$age > 24 & stat$age <= 30), "Middle Aged", "Young")))

 

stat$agecat

stat

 

 

# or more compactly

 

stat <- within(stat, {

  agecat1 <- NA

  agecat1[age > 30] <- "Old"

  agecat1[age >= 25 & age <= 30] <- "Middle Aged"

  agecat1[age < 25] <- "Young"

})

 

stat

 

class(stat$agecat)

 

stat$agecat <- factor(stat$agecat, levels = c("Old", "Middle Aged", "Young"))

stat

table(stat$agecat)  # table drops NA by default

table(stat$agecat, useNA = "ifany")

?table

 

 

# Renaming variables with the reshape package

install.packages("reshape")

library(reshape)

 

colnames(stat)

 

stat <- rename(stat, c(squad = "squadID", date="testdate"))

 

stat

 

# or rename variables without using the reshape package

 

names(stat)

names(stat)[1] <- "squadID"

names(stat)[2] <- "testdate"

names(stat)[5:9] <- paste("field", 1:5, sqp="")

 

y <- c(1,2,3,NA)

is.na(y)

 

 

# Applying the is.na() function

is.na(stat[,5:9])

complete.cases(stat[,5:9])

 

 

# recode 97 to missing for the variable age

stat$age[stat$age == 97] <- NA

stat

 

# na.rm=TRUE option

x <- c(1,2,NA,3)

x

y <- sum(x); y

y <- sum(x, na.rm=TRUE); y

x

 

 

# Using na.omit() to delete incomplete observations

stat

newdata <- na.omit(stat)

newdata

경축! 아무것도 안하여 에스천사게임즈가 새로운 모습으로 재오픈 하였습니다.
어린이용이며, 설치가 필요없는 브라우저 게임입니다.
https://s1004games.com

complete.cases(stat)

stat[complete.cases(stat),]

 

# default format for inputting dates : yyyy-mm-dd

mydates <- as.Date(c("2016-01-27", "2020-02-13")); mydates

class(mydates)

 

# Converting character values to dates

 

strDates <- c("01/05/2000", "08/16/2000")

dates <- as.Date(strDates, "%m/%d/%Y"); dates

 

 

install.packages("chron")

library(chron)

Sys.Date()

date <- seq(Sys.Date(), by = 1, length.out = 7)

date

 

# Useful date functions

 

Sys.Date()

date()

 

# Use the format function to output dates in a special format,

# and to extract portions of dates

today <- Sys.Date()

today

format(today, format = "%B %d %Y")

today

format(today, format = "%A")

today

 

# Arithmetic operations on dates

 

startdate <- as.Date("2016-02-10")

enddate <- as.Date("2017-12-12")

days <- enddate - startdate; days

as.numeric(days)

difftime(enddate, startdate)

# difftime(Sys.Date()+ 1, as.Date("2016-05-07"))

 

# difftime function

# How old am I?

today <- Sys.Date()

dob <- as.Date("2010-10-10")

difftime(today,dob)

difftime(today, dob, units="weeks")

difftime(today, dob, units="days") # default

 

 

# converting dates to character variables

class(dates)

strDates <- as.character(dates)

strDates

class(strDates)

 

# Converting from one date type to another

 

a <- c(1,2,3)

a

is.numeric(a)

is.vector(a)

a <- as.character(a)

a

 

is.numeric(a)

is.vector(a)

is.character(a)

 

 

#### sort,order ####

# sorting a dataset 

 

# create a new dataset containing rows sorted from youngest to oldset squad

newdata <- stat[order(stat$age),]

newdata$age

 

# sort the rows into female follewed by male,

# and youngest to oldest within each gender.

attach(stat)

newdata <- stat[order(gender, -age), ]

newdata

detach()

 

 

# Selecting variables

 

newdata <- stat[, c(5:9)]; newdata

variables <- paste("field", 1:5, sep=""); variables

newdata <- stat[variables]; newdata

 

# Dropping variables

names(stat)[5:9] <- paste("field", 1:5, sep="")

names(stat)

variables <- names(stat) %in% c("field3", "field4"); variables

newdata <- stat[!variables]; newdata

 

# or use column numbers to drop variables

t(colnames(stat))

newdata <- stat[-c(8,9)]; newdata

 

 

# You could use the following to delete f3 and f4

# from the stat dataset (commented out so 

# the rest of the code in this file will work)

#

# stat$f3 <- stat$f4 <- NULL

 

newdata <- stat[1:3,]; newdata

 

newdata <- stat[which(stat$gender == "M" &

                        stat$age > 30),]

 

newdata

 

attach(stat)

newdata <- stat[which(gender == "M" & age > 30),]

detach(stat)

 

 

newdata <- subset(stat, gender=="M" & age > 30); newdata

 

# Selecting observations based on dates

 

stat$date <- as.Date(stat$date, "%m/%d/%Y");

startdate <- as.Date("2017-02-01")

enddate <- as.Date("2017-02-28")

newdata <- stat[stat$date >= startdate & 

                  stat$date <= enddate,]

 

newdata

 

# Using the subset() function

 

newdata <- subset(stat, age >= 35 | age < 24, 

                  select = c(field1, field2, field3, field4 ))

 

newdata <- subset(stat, gender=="M" & age > 25,

                  select = gender:field4)

 

newdata

 

 

 

본 웹사이트는 광고를 포함하고 있습니다.
광고 클릭에서 발생하는 수익금은 모두 웹사이트 서버의 유지 및 관리, 그리고 기술 콘텐츠 향상을 위해 쓰여집니다.
번호 제목 글쓴이 날짜 조회 수
공지 오라클 기본 샘플 데이터베이스 졸리운_곰 2014.01.02 86157
공지 [SQL컨셉] 서적 "SQL컨셉"의 샘플 데이타 베이스 SAMPLE DATABASE of ORACLE 가을의 곰을... 2013.02.10 78652
공지 [G_SQL] Sample Database 가을의 곰을... 2012.05.20 95385
9 [암호화폐] [파이썬] 암호화폐 자동매매(1): 변동성 전략 +상승장 졸리운_곰 2025.03.13 1809
8 [암호화폐] Solana 토큰 만들기 — MeMe Coin file 졸리운_곰 2024.11.15 1244
7 [암호화폐] 솔리디티를 이용해 이더리움 스마트 계약 시작하기 file 졸리운_곰 2024.04.05 1736
6 암호화폐 (비트코인, cryptocurrency, bitcoin) 파이썬을 이용한 가상화폐 시세 분석 file 졸리운_곰 2024.03.28 1987
5 암호화폐 (비트코인, cryptocurrency, bitcoin) Solidity 이더리움 Solidity Tutorial: How to build and deploy a smart contract to send Ether from one account to another file 졸리운_곰 2024.01.23 1182
4 암호화폐 (비트코인, cryptocurrency, bitcoin) Solidity 이더리움 Cheatsheet 졸리운_곰 2024.01.23 1761
3 암호화폐 (비트코인, cryptocurrency, bitcoin) [Ethereum] Remix 를 이용하여 이더리움 솔리디티(Solidity) 개발 연습 하기! file 졸리운_곰 2021.10.19 1388
2 암호화폐 (비트코인, cryptocurrency, bitcoin) [Ethereum] Remix IDE를 이용한 Solidity 프로그래밍 file 졸리운_곰 2021.10.17 1794
1 암호화폐 (비트코인, cryptocurrency, bitcoin) [Ethereum] 스마트 컨트렉트로 "Hello, World"를 출력하자.​ file 졸리운_곰 2021.10.09 2095
대표 김성준 주소 : 경기 용인 분당수지 U타워 등록번호 : 142-07-27414
통신판매업 신고 : 제2012-용인수지-0185호 출판업 신고 : 수지구청 제 123호 개인정보보호최고책임자 : 김성준 sjkim70@stechstar.com
대표전화 : 010-4589-2193 [fax] 02-6280-1294 COPYRIGHT(C) stechstar.com ALL RIGHTS RESERVED