#------------------ # Data Preparation #------------------ #Read datasets #Download the data from http://www.saedsayad.com/datasets/CreditData.zip train <- read.csv("Credit_train.csv") test <- read.csv("Credit_test.csv") #Rows and Cols dim(train) dim(test) #Columns name colnames(train) colnames(test) #Show head(train) head(test) #---------------------------------------------------- # Data Exploration - Univariate analysis - Numerical #---------------------------------------------------- #Statistics summary(train) summary(test) # MAXLINEUTIL boxplot(train$MAXLINEUTIL, main="Maximum number of lines being utilized", sub="train" , xlab="MAXLINEUTIL", col="darkgreen") boxplot(test$MAXLINEUTIL, main="Maximum number of lines being utilized", sub="test" , xlab="MAXLINEUTIL", col="brown") hist(train$MAXLINEUTIL, main="Maximum number of lines being utilized", sub="train" , xlab="MAXLINEUTIL", breaks=50, col="darkgreen") hist(test$MAXLINEUTIL, main="Maximum number of lines being utilized", sub="test" , xlab="MAXLINEUTIL", breaks=50, col="brown") # DAYSDELQ boxplot(train$DAYSDELQ, main="Number of delinquent days", sub="train" , xlab="DAYSDELQ", col="darkgreen") boxplot(test$MAXLINEUTIL, main="Number of delinquent days", sub="test" , xlab="DAYSDELQ", col="brown") hist(train$DAYSDELQ, main="Number of delinquent days", sub="train" , xlab="DAYSDELQ", breaks=50, col="darkgreen") hist(test$DAYSDELQ, main="Number of delinquent days", sub="test" , xlab="DAYSDELQ", breaks=50, col="brown") # TOTACBAL boxplot(train$TOTACBAL, main="Total balance of business account", sub="train" , xlab="TOTACBAL", col="darkgreen") boxplot(test$TOTACBAL, main="Total balance of business account", sub="test" , xlab="TOTACBAL", col="brown") hist(train$TOTACBAL, main="Total balance of business account", sub="train" , xlab="TOTACBAL", breaks=50, col="darkgreen") hist(test$TOTACBAL, main="Total balance of business account", sub="test" , xlab="TOTACBAL", breaks=50, col="brown")