-
Notifications
You must be signed in to change notification settings - Fork 1
Expand file tree
/
Copy pathrun_pipeline.sh
More file actions
104 lines (91 loc) 路 3.45 KB
/
Copy pathrun_pipeline.sh
File metadata and controls
104 lines (91 loc) 路 3.45 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
#!/bin/bash
# ----------------------------------- Step 1 -----------------------------------
python ./preprocess.py \
--txt-path './datasets/GSE13355_series_matrix.txt' \
--csv-path './datasets/GSE13355.csv' \
--pkl-path './datasets/GSE13355.pkl' \
--target-header '!Sample_characteristics_ch1' \
--target-regexs '^involved.*' '^.*controls.*$' \
--new-targets 'Psoriasis' 'Normal'
python ./preprocess.py \
--txt-path './datasets/GSE14905_series_matrix.txt' \
--csv-path './datasets/GSE14905.csv' \
--pkl-path './datasets/GSE14905.pkl' \
--target-header '!Sample_characteristics_ch1' \
--target-regexs '^.*psoriasis.*$' '^.*normal.*$' \
--new-targets 'Psoriasis' 'Normal'
# NOTE: if you want to integrate datasets from different platforms (not needed here), you have to:
# 1) annotate the dataset
Rscript annotate.R 'GSE13355' 'characteristics_ch1' './datasets/GSE13355annotated.csv'
# 2) pre-process the annotated dataset
python ./preprocess_annotated.py \
--dataset 'GSE13355annotated.csv' \
--target-regexs '^involved.*' '^.*controls.*$' \
--new-targets 'Psoriasis' 'Normal' \
--pkl-path './datasets/GSE13355.pkl'
# --------------------------------- Step 1.2 -----------------------------------
python ./integrate.py \
--pkl-in './datasets/GSE13355.pkl' './datasets/GSE14905.pkl' \
--pkl-out './datasets/GSE13355-GSE14905.pkl'
python ./integrate.py \
--pkl-in './datasets/GSE13355.pkl' './datasets/GSE14905.pkl' \
--pkl-out './datasets/GSE13355_sub.pkl' './datasets/GSE14905_sub.pkl'
# ----------------------------------- Step 2 -----------------------------------
# Create folders to store results
mkdir './results/'
mkdir './results/GSE13355/'
mkdir './results/GSE14905/'
mkdir './results/GSE13355-GSE14905/'
mkdir './results/GSE13355-TR-GSE14905-TS/'
mkdir './results/GSE13355/feature_importance/'
mkdir './results/GSE13355/feature_elimination/'
# Compare classifiers on GSE13355
python ./compare.py \
--nested-cv \
--dev-dataset './datasets/GSE13355.pkl' \
--output-path './results/GSE13355/' \
--best-score 'concat f1 weighted' \
--n-splits 5 \
--ext-n-splits 5 \
--verbose
# Compare classifiers on GSE14905
python ./compare.py \
--nested-cv \
--dev-dataset './datasets/GSE14905.pkl' \
--output-path './results/GSE14905/' \
--best-score 'concat f1 weighted' \
--n-splits 5 \
--ext-n-splits 5 \
--verbose
# Compare classifiers on the dataset obtained merging GSE13355 and GSE14905
python ./compare.py \
--nested-cv \
--dev-dataset './datasets/GSE13355-GSE14905.pkl' \
--output-path './results/GSE13355-GSE14905/' \
--best-score 'concat f1 weighted' \
--n-splits 5 \
--ext-n-splits 5 \
--verbose
# Compare classifiers training them on GSE13355 and testing them on GSE14905
python ./compare.py \
--dev-dataset './datasets/GSE13355_sub.pkl' \
--test-dataset './datasets/GSE14905_sub.pkl' \
--output-path './results/GSE13355-TR-GSE14905-TS/' \
--best-score 'concat f1 weighted' \
--n-splits 5 \
--ext-n-splits 5 \
--verbose
# SVM default parameters with linear kernel
echo '{"kernel" : "linear"}' > './results/GSE13355/SVM_params.json'
# Apply Recursive Feature Elimination on GSE13355
python ./RFE.py \
--dataset './datasets/GSE13355.pkl' \
--rac-params './results/GSE13355-TR-GSE14905-TS/RAC_params.json' \
--svm-params './results/GSE13355/SVM_params.json' \
--rf-params './results/GSE13355-TR-GSE14905-TS/RF_params.json' \
--xgb-params './results/GSE13355-TR-GSE14905-TS/XGB_params.json' \
--output-path './results/GSE13355/feature_elimination/' \
--n-splits 5 \
--n-features-to-select 20 \
--step 0.5 \
--verbose