-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathTrainingModel.R
More file actions
142 lines (108 loc) · 5.29 KB
/
Copy pathTrainingModel.R
File metadata and controls
142 lines (108 loc) · 5.29 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
# Load dataset
pima_data <- read.csv("data/diabetes.csv", colClasses = c(
Pregnancies = "numeric",
Glucose = "numeric",
BloodPressure = "numeric",
SkinThickness = "numeric",
Insulin = "numeric",
BMI = "numeric",
DiabetesPedigreeFunction = "numeric",
Age = "numeric",
Outcome = "factor"
), header = TRUE)
# Display the structure of the dataset
str(pima_data)
# View the first few rows of the dataset
head(pima_data)
# Open the dataset in a viewer window
View(pima_data)
# Load necessary libraries
library(caret)
# Set seed for reproducibility
set.seed(123)
# Split the data into 70% training and 30% testing
train_indices <- createDataPartition(pima_data$Outcome, p = 0.7, list = FALSE)
# Create training and testing sets
train_data <- pima_data[train_indices, ]
test_data <- pima_data[-train_indices, ]
# Display the dimensions of the training and testing sets
cat("Training data dimensions:", dim(train_data), "\n")
cat("Testing data dimensions:", dim(test_data), "\n")
# Load necessary libraries
library(boot)
# Define the function to compute the statistic of interest (mean glucose level)
compute_statistic <- function(data, indices) {
sample_data <- data[indices, ]
mean_glucose <- mean(sample_data$Glucose, na.rm = TRUE)
return(mean_glucose)
}
# Set the number of bootstrap replicates
num_replicates <- 1000
# Perform bootstrapping
bootstrapped_means <- boot(data = pima_data, statistic = compute_statistic, R = num_replicates)
# Display the bootstrapped mean glucose levels
print(bootstrapped_means)
# Load necessary libraries
library(caret)
# Define the control parameters for cross-validation
ctrl <- trainControl(method = "cv", # Use k-fold cross-validation
number = 10) # Specify the number of folds (e.g., 10-fold)
# Define the predictive model (e.g., linear regression)
model <- train(Outcome ~ ., # Specify the formula for the model
data = pima_data, # Specify the dataset
method = "glm", # Specify the modeling method (e.g., generalized linear model)
trControl = ctrl) # Specify the control parameters for cross-validation
# Display the cross-validation results
print(model)
# Load necessary libraries
library(caret)
# Define the control parameters for model training
ctrl <- trainControl(method = "cv", # Use k-fold cross-validation
number = 10) # Specify the number of folds (e.g., 10-fold)
# Train logistic regression model
logistic_model <- train(Outcome ~ ., # Specify the formula for the model
data = pima_data, # Specify the dataset
method = "glm", # Specify the modeling method (logistic regression)
trControl = ctrl) # Specify the control parameters for cross-validation
# Display the trained logistic regression model
print(logistic_model)
# Train decision tree model
decision_tree_model <- train(Outcome ~ ., # Specify the formula for the model
data = pima_data, # Specify the dataset
method = "rpart", # Specify the modeling method (decision trees)
trControl = ctrl) # Specify the control parameters for cross-validation
# Display the trained decision tree model
print(decision_tree_model)
# Train SVM model
svm_model <- train(Outcome ~ ., # Specify the formula for the model
data = pima_data, # Specify the dataset
method = "svmRadial", # Specify the modeling method (SVM with radial kernel)
trControl = ctrl) # Specify the control parameters for cross-validation
# Display the trained SVM model
print(svm_model)
# Load necessary libraries
library(caret)
# Define the control parameters for model training
ctrl <- trainControl(method = "cv", # Use k-fold cross-validation
number = 10) # Specify the number of folds (e.g., 10-fold)
# Train logistic regression model
logistic_model <- train(Outcome ~ ., # Specify the formula for the model
data = pima_data, # Specify the dataset
method = "glm", # Specify the modeling method (logistic regression)
trControl = ctrl) # Specify the control parameters for cross-validation
# Train decision tree model
decision_tree_model <- train(Outcome ~ ., # Specify the formula for the model
data = pima_data, # Specify the dataset
method = "rpart", # Specify the modeling method (decision trees)
trControl = ctrl) # Specify the control parameters for cross-validation
# Train SVM model
svm_model <- train(Outcome ~ ., # Specify the formula for the model
data = pima_data, # Specify the dataset
method = "svmRadial", # Specify the modeling method (SVM with radial kernel)
trControl = ctrl) # Specify the control parameters for cross-validation
# Compare model performance using resamples
model_results <- resamples(list(Logistic = logistic_model,
Decision_Tree = decision_tree_model,
SVM = svm_model))
# Summarize the model performance
summary(model_results)