-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathMakefile
More file actions
202 lines (183 loc) · 8.02 KB
/
Copy pathMakefile
File metadata and controls
202 lines (183 loc) · 8.02 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
# ===================================================================
# MIND Training Makefile
# ===================================================================
# Usage:
# make train GPU=6 CONFIG=m # multidomain, default log
# make train GPU=6 CONFIG=f # fast test (~10 min)
# make train GPU=6 CONFIG=m LOG=my_run # custom log name
# make train GPU=auto CONFIG=f # auto-pick least used GPU
#
# CONFIG shortcuts:
# m = pretraining_config_multidomain.yaml
# f = pretraining_config_fast.yaml
# (or pass full filename without .yaml)
#
# Defaults: GPU=auto, CONFIG=m, LOG=train_<config>
# ===================================================================
SHELL := /bin/bash
CONDA_RUN := source /opt/anaconda3/bin/activate && conda activate mind_esa3d &&
GPU ?= auto
CONFIG ?= m
LOG ?=
EPOCHS ?=
SEED ?=
PYTEST_ARGS ?= tests --ignore=tests/test_multigpu_training.py
ifneq ($(strip $(EPOCHS)),)
TRAIN_EPOCH_ARGS := --max-epochs $(EPOCHS)
endif
ifneq ($(strip $(SEED)),)
TRAIN_SEED_ARGS := --seed $(SEED)
endif
# Map single-letter shortcuts to full config names
ifeq ($(CONFIG),m)
CONFIG_FILE := core/pretraining_config_multidomain.yaml
CONFIG_TAG := multidomain
CONFIG_IS_FAST := 0
else ifeq ($(CONFIG),f)
CONFIG_FILE := core/pretraining_config_fast.yaml
CONFIG_TAG := fast
CONFIG_IS_FAST := 1
else
CONFIG_FILE := core/$(CONFIG).yaml
CONFIG_TAG := $(CONFIG)
CONFIG_IS_FAST := 0
endif
# Log directory: logs/<MonthYear>/ e.g. logs/June2026/
LOG_MONTH_DIR := logs/$(shell date +%B%Y)
# Log filename: <name><datetime>.log or <name>fast<datetime>.log
_LOG_DATETIME := $(shell date +%Y%m%d_%H%M%S)
ifeq ($(LOG),)
_LOG_BASE := $(CONFIG_TAG)
else
_LOG_BASE := $(LOG)
endif
ifeq ($(CONFIG_IS_FAST),1)
LOG_FILE := $(LOG_MONTH_DIR)/$(_LOG_BASE)fast_$(_LOG_DATETIME).log
else
LOG_FILE := $(LOG_MONTH_DIR)/$(_LOG_BASE)_$(_LOG_DATETIME).log
endif
# A unique checkpoint directory is essential when independent validation runs
# share the machine. The old single default directory allowed concurrent jobs
# to overwrite each other's last.ckpt.
OUTPUT_DIR ?= outputs/$(_LOG_BASE)_$(_LOG_DATETIME)
MIN_VRAM_MB ?= 15000
# Resolve GPU: if "auto", pick GPU with most free memory (must have >= MIN_VRAM_MB free)
RESOLVED_GPU = $(shell if [ "$(GPU)" = "auto" ]; then \
best=$$(nvidia-smi --query-gpu=index,memory.free --format=csv,noheader,nounits 2>/dev/null | \
awk -F', ' '$$2 >= $(MIN_VRAM_MB) {print $$1, $$2}' | sort -k2 -rn | head -1 | cut -d' ' -f1); \
if [ -z "$$best" ]; then echo "NONE"; else echo "$$best"; fi; \
else echo "$(GPU)"; fi)
# ===================================================================
# Data Processing
# ===================================================================
# make process-protein # sequential (all 10 chunks)
# make process-protein PARALLEL=2 # 2 parallel groups (nohup)
# ===================================================================
NUM_CHUNKS ?= 10
PARALLEL ?= 1
PROCESS_ARGS = --config-yaml-path core/pretraining_config_multidomain.yaml \
--dataset pdb \
--data-path /home/yusuf/data/proteins/raw_structures_hq_40k \
--manifest-file /home/yusuf/data/proteins/afdb_clusters/manifest_hq_40k_len512.csv \
--output-base /home/yusuf/data/proteins/processed_14atom \
--cache-dir /home/yusuf/data/proteins/cache_14atom \
--num-chunks $(NUM_CHUNKS) \
--use-hybrid-edges --force
.PHONY: train test-unit stop status logs help process-protein
test-unit:
$(CONDA_RUN) python -m pytest -q $(PYTEST_ARGS)
train:
@if [ "$(RESOLVED_GPU)" = "NONE" ]; then \
echo "ERROR: No GPU with >= $(MIN_VRAM_MB) MB free VRAM found."; \
echo "Available GPUs:"; \
nvidia-smi --query-gpu=index,memory.free,memory.total --format=csv,noheader 2>/dev/null; \
echo ""; \
echo "Use GPU=<number> to force a specific GPU, or wait for one to free up."; \
exit 1; \
fi
@echo "=== MIND Training ==="
@echo " GPU: $(RESOLVED_GPU) $(if $(filter auto,$(GPU)),(auto-selected))"
@echo " Config: $(CONFIG_FILE)"
@echo " Epochs: $(if $(strip $(EPOCHS)),$(EPOCHS),config default)"
@echo " Seed: $(if $(strip $(SEED)),$(SEED),config default)"
@echo " Log: $(LOG_FILE)"
@echo " Output: $(OUTPUT_DIR)"
@echo "====================="
@mkdir -p $(LOG_MONTH_DIR)
$(CONDA_RUN) CUDA_VISIBLE_DEVICES=$(RESOLVED_GPU) nohup python -m core.train_pretrain \
--config-yaml-path $(CONFIG_FILE) \
$(TRAIN_EPOCH_ARGS) \
$(TRAIN_SEED_ARGS) \
--output-dir $(OUTPUT_DIR) \
> $(LOG_FILE) 2>&1 &
@echo "PID: $$!"
@echo "Follow: tail -f $(LOG_FILE)"
stop:
@echo "Killing training processes on GPU $(GPU)..."
@ps aux | grep "CUDA_VISIBLE_DEVICES=$(GPU).*train_pretrain" | grep -v grep | awk '{print $$2}' | xargs -r kill
@echo "Done."
status:
@echo "=== MIND Training (all users) ==="
@ps -eo user:12,pid,etime,args | grep '[t]rain_pretrain\|core[./]train' | grep -v grep || echo " (no MIND training running)"
@echo ""
@echo "=== MIND Data Processing (all users) ==="
@ps -eo user:12,pid,args | grep '[c]ache_to_pyg\|create_chunk' | grep -v grep | \
awk '{print $$1}' | sort | uniq -c | \
awk '{printf " %s: %d worker processes\n", $$2, $$1}' || echo " (none)"
@echo ""
@echo "=== GPU Usage ==="
@nvidia-smi --query-gpu=index,memory.used,memory.total,utilization.gpu --format=csv,noheader 2>/dev/null || true
@echo ""
@echo "=== GPU Processes ==="
@nvidia-smi --query-compute-apps=gpu_name,pid,used_gpu_memory,process_name --format=csv,noheader 2>/dev/null | \
while IFS=, read gpu pid mem name; do \
usr=$$(ps -o user= -p $$(echo $$pid | tr -d ' ') 2>/dev/null | tr -d ' '); \
idx=$$(nvidia-smi --query-compute-apps=gpu_bus_id,pid --format=csv,noheader 2>/dev/null | \
grep "$$(echo $$pid | tr -d ' ')" | head -1 | cut -d, -f1 | tr -d ' '); \
gpuidx=$$(nvidia-smi --query-gpu=index,gpu_bus_id --format=csv,noheader 2>/dev/null | \
grep "$$idx" | head -1 | cut -d, -f1 | tr -d ' '); \
printf " GPU %-2s | %-14s | PID %-8s |%s\n" "$${gpuidx:-?}" "$${usr:-?}" "$$pid" "$$mem"; \
done 2>/dev/null || true
process-protein:
ifeq ($(PARALLEL),1)
@echo "=== Processing protein data (sequential, all $(NUM_CHUNKS) chunks) ==="
$(CONDA_RUN) nohup python data_loading/process_chunked_dataset.py \
$(PROCESS_ARGS) > process_protein.log 2>&1 &
@echo "PID: $$!"
@echo "Follow: tail -f process_protein.log"
else
@echo "=== Processing protein data ($(PARALLEL) parallel groups) ==="
@chunks_per_group=$$(( ($(NUM_CHUNKS) + $(PARALLEL) - 1) / $(PARALLEL) )); \
for g in $$(seq 0 $$(($(PARALLEL)-1))); do \
start=$$((g * chunks_per_group)); \
end_idx=$$((start + chunks_per_group - 1)); \
if [ $$end_idx -ge $(NUM_CHUNKS) ]; then end_idx=$$(($(NUM_CHUNKS)-1)); fi; \
if [ $$start -ge $(NUM_CHUNKS) ]; then continue; fi; \
echo " Group $$g: chunks $$start-$$end_idx -> process_protein_$${g}.log"; \
$(CONDA_RUN) nohup python data_loading/process_chunked_dataset.py \
$(PROCESS_ARGS) --chunk-range "$$start-$$end_idx" \
> process_protein_$${g}.log 2>&1 & \
done
@echo ""
@echo "Follow: tail -f process_protein_*.log"
endif
logs:
@find logs/ -name "*.log" -printf "%T@ %p\n" 2>/dev/null | sort -rn | head -10 | awk '{print $$2}' || echo "No log files found."
help:
@echo "MIND Commands:"
@echo ""
@echo " Training:"
@echo " make train CONFIG=m LOG=run1 Full multidomain training"
@echo " make train CONFIG=f Fast sanity check (~10 min)"
@echo " make train CONFIG=f EPOCHS=1 Override config epoch count"
@echo " make train CONFIG=f SEED=43 Independent seeded run"
@echo " make test-unit Run the complete unit-test suite"
@echo " make stop GPU=6 Kill training on GPU 6"
@echo " make status Show all processes + GPUs"
@echo " make logs List recent log files"
@echo ""
@echo " Data Processing:"
@echo " make process-protein All chunks (sequential, nohup)"
@echo " make process-protein PARALLEL=2 2 parallel groups (nohup)"
@echo ""
@echo " CONFIG shortcuts: m=multidomain, f=fast"