# read in the data from github repo:
income <- read.csv("https://raw.githubusercontent.com/selva86/datasets/master/income.csv")

set.seed(100)

# We shuffle row-wise:
incomeR <- income[sample(nrow(income)),]

#check rownames (see above screenshot)
colnames(incomeR)

# Here we replace NAs
incomeR <- incomeR %>%
mutate_if(is.factor, fct_explicit_na, na_level = 'Unknown') %>%
mutate(INCOME = as.factor(INCOME))

#install packages
install.packages('randomForest')
library(randomForest)
install.packages('ggRandomForests')
library(ggRandomForests)

# When running this bear in mind that it could take a minute or two
model_base <- randomForest(INCOME ~ ., data = incomeR, importance = TRUE)