Descarga datos desde: Dataset Defaulteo de Credito
1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 |
#------------------ # Preparacion de Datos #------------------ #Leer datasets #Descargar los datos desde: http://datascience.esy.es/wp-content/uploads/2018/03/CreditData-1.zip train <- read.csv("Credit_train.csv") test <- read.csv("Credit_test.csv") #Lineas y Columnas dim(train) dim(test) #Nombre de Columnas colnames(train) colnames(test) #Mostrar head(train) head(test) #---------------------------------------------------- # Exploracion de datos - analisis de univariable - Numerico #---------------------------------------------------- #Estadisticas summary(train) summary(test) # MAXLINEUTIL boxplot(train$MAXLINEUTIL, main="Maximum number of lines being utilized", sub="train" , xlab="MAXLINEUTIL", col="darkgreen") boxplot(test$MAXLINEUTIL, main="Maximum number of lines being utilized", sub="test" , xlab="MAXLINEUTIL", col="brown") hist(train$MAXLINEUTIL, main="Maximum number of lines being utilized", sub="train" , xlab="MAXLINEUTIL", breaks=50, col="darkgreen") hist(test$MAXLINEUTIL, main="Maximum number of lines being utilized", sub="test" , xlab="MAXLINEUTIL", breaks=50, col="brown") # DAYSDELQ boxplot(train$DAYSDELQ, main="Number of delinquent days", sub="train" , xlab="DAYSDELQ", col="darkgreen") boxplot(test$MAXLINEUTIL, main="Number of delinquent days", sub="test" , xlab="DAYSDELQ", col="brown") hist(train$DAYSDELQ, main="Number of delinquent days", sub="train" , xlab="DAYSDELQ", breaks=50, col="darkgreen") hist(test$DAYSDELQ, main="Number of delinquent days", sub="test" , xlab="DAYSDELQ", breaks=50, col="brown") # TOTACBAL boxplot(train$TOTACBAL, main="Total balance of business account", sub="train" , xlab="TOTACBAL", col="darkgreen") boxplot(test$TOTACBAL, main="Total balance of business account", sub="test" , xlab="TOTACBAL", col="brown") hist(train$TOTACBAL, main="Total balance of business account", sub="train" , xlab="TOTACBAL", breaks=50, col="darkgreen") hist(test$TOTACBAL, main="Total balance of business account", sub="test" , xlab="TOTACBAL", breaks=50, col="brown") |