-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathremote_sshd.sh
More file actions
executable file
·252 lines (231 loc) · 6.97 KB
/
Copy pathremote_sshd.sh
File metadata and controls
executable file
·252 lines (231 loc) · 6.97 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
#!/bin/bash
current_path="$(dirname "$0")"
source "$current_path/cluster_helpers.sh"
# Set default value.
CLUSTER=bluehive3
PARTITION=doppelbock
CPUS=16
GPUS=1
MEMORY=256
TIME=24
PORT=30022
ROOT_OVERRIDE=""
usage() {
echo "Usage: $0 [-a CLUSTER] [-p PARTITION] [-c CPUS] [-g GPUS] [-m MEMORY_GB] [-t HOURS] [-w NODE] [--root PATH]"
echo "Supported clusters: $(cluster_supported_list)"
}
while [[ $# -gt 0 ]]; do
case "$1" in
-p|--partition)
if [[ -z "$2" || "$2" == -* ]]; then
echo "Error: $1 requires a partition" >&2
exit 1
fi
PARTITION=$2
shift 2
;;
-a|--cluster)
if [[ -z "$2" || "$2" == -* ]]; then
echo "Error: $1 requires a cluster name" >&2
exit 1
fi
CLUSTER=$2
shift 2
;;
--cluster=*)
CLUSTER="${1#*=}"
shift
;;
-c|--cpus)
if [[ -z "$2" || "$2" == -* ]]; then
echo "Error: $1 requires a CPU count" >&2
exit 1
fi
CPUS=$2
shift 2
;;
-g|--gpus)
if [[ -z "$2" || "$2" == -* ]]; then
echo "Error: $1 requires a GPU count" >&2
exit 1
fi
GPUS=$2
shift 2
;;
-m|--memory)
if [[ -z "$2" || "$2" == -* ]]; then
echo "Error: $1 requires a memory value" >&2
exit 1
fi
MEMORY=$2
shift 2
;;
-t|--time)
if [[ -z "$2" || "$2" == -* ]]; then
echo "Error: $1 requires a runtime in hours" >&2
exit 1
fi
TIME=$2
shift 2
;;
-w|--node)
if [[ -z "$2" || "$2" == -* ]]; then
echo "Error: $1 requires a node name" >&2
exit 1
fi
NODE=$2
shift 2
;;
-r|--root)
if [[ -z "$2" || "$2" == -* ]]; then
echo "Error: $1 requires a remote path" >&2
exit 1
fi
ROOT_OVERRIDE="$2"
shift 2
;;
--root=*)
ROOT_OVERRIDE="${1#*=}"
shift
;;
-h|--help)
usage
exit 0
;;
--)
shift
break
;;
*)
echo "Error: Unexpected argument '$1'" >&2
usage >&2
exit 1
;;
esac
done
echo "CLUSTER: $CLUSTER"
echo "PARTITION: $PARTITION"
echo "CPUS: $CPUS"
echo "GPUS: $GPUS"
echo "MEMORY: $MEMORY"
echo "TIME: $TIME"
echo "NODE: $NODE"
HOSTNAME="$(cluster_hostname "$CLUSTER")" || exit 1
source "$current_path/start_ssh_control.sh" -a "$CLUSTER"
if [ -n "$ROOT_OVERRIDE" ]; then
REMOTE_SHARED_ROOT="$ROOT_OVERRIDE"
export REMOTE_SHARED_ROOT
fi
echo "REMOTE_SHARED_ROOT: $REMOTE_SHARED_ROOT"
source "$current_path/remote_tools.sh"
ensure_remote_dropbear || exit $?
DROPBEAR_DIR="$REMOTE_SHARED_ROOT/dropbear"
ssh -F /dev/null -o ControlMaster=auto \
-o "ControlPath=$SSH_CONTROL_PATH" -o StrictHostKeyChecking=no \
-T "$SSH_LOGIN_TARGET" <<ENDSSH
#!/bin/bash
module load gcc 2>/dev/null || true
mkdir -p /home/$USER/logs
# Your commands go here
if squeue -u $USER -O name:32|grep my_sshd; then
echo "SSHD is already running."
job=\$(squeue -u $USER -O jobarrayid:18,partition:13,username:12,submittime:22,starttime:22,timeused:13,timelimit:13,numcpus:10,gres:15,minmemory:12,nodelist:10,priorityLong:9,reason:9,name:4)
echo "\$job"
else
rm -rf /home/$USER/logs/dropbear.log
cat <<'INNEREOF' | sbatch
#!/bin/bash
#SBATCH -p $PARTITION -t $TIME:00:00
#SBATCH -c $CPUS
#SBATCH --mem="${MEMORY}G"
#SBATCH --gres=gpu:$GPUS
#SBATCH -o /home/$USER/logs/dropbear.log
#SBATCH --job-name=my_sshd
cd "$DROPBEAR_DIR"
# Function to check if a port is available using ss command
function check_port_available() {
local port=\$1
# Check if the port is in use
ss -tuln | grep ":\$port " > /dev/null
if [ \$? -ne 0 ]; then
return 0 # Port is available
else
return 1 # Port is in use
fi
}
# Find an available port in range 30000-40000
function find_available_port() {
local port
# Try up to 50 times to find an available port
for attempt in {1..50}; do
# Generate a random port number between 30000 and 40000
port=\$((30000 + RANDOM % 10000))
if check_port_available \$port; then
echo \$port
return 0
fi
done
# If no port found after 50 attempts, use default 30022
echo 30022
return 1
}
# Get an available port
PORT=\$(find_available_port)
echo "Using port: \$PORT"
echo "Using Node: \$SLURM_JOB_NODELIST"
# Start dropbear with the selected port
./sbin/dropbear -F -E -p \$PORT -r ./.ssh/dropbear_rsa_host_key -r ./.ssh/dropbear_ecdsa_host_key -r ./.ssh/dropbear_ed25519_host_key
INNEREOF
fi
ENDSSH
# SSH into cluster and check for port in a loop
PORT=$(ssh -F /dev/null -o ControlMaster=auto \
-o "ControlPath=$SSH_CONTROL_PATH" -o StrictHostKeyChecking=no \
-T "$SSH_LOGIN_TARGET" <<ENDSSH2
while true; do
# Get SSH port from the job if it's running
PORT_INFO=\$(grep 'Using port:' /home/$USER/logs/dropbear.log 2>/dev/null | tail -1)
PORT=\$(echo \$PORT_INFO | awk '{print \$3}')
# Break the loop if a valid port (greater than 0) is found
if [ -n "\$PORT" ]; then
echo \$PORT
break
fi
sleep 1
done
ENDSSH2
)
NODE=$(ssh -F /dev/null -o ControlMaster=auto \
-o "ControlPath=$SSH_CONTROL_PATH" -o StrictHostKeyChecking=no \
-T "$SSH_LOGIN_TARGET" <<ENDSSH2
echo "Waiting for port allocation..." > /home/$USER/logs/dropbear_test.log
while true; do
# Get SSH port from the job if it's running
NODE_INFO=\$(grep 'Using Node:' /home/$USER/logs/dropbear.log 2>/dev/null | tail -1)
NODE=\$(echo \$NODE_INFO | awk '{print \$3}')
# Break the loop if a valid port (greater than 0) is found
if [ -n "\$NODE" ]; then
echo \$NODE
break
fi
sleep 1
done
ENDSSH2
)
echo "Detected port: $PORT"
echo "Detected node: $NODE"
remote_sshd_state_directory="${XDG_STATE_HOME:-${HOME}/.local/state}/unix-scripts"
remote_sshd_state_file="${remote_sshd_state_directory}/remote_sshd_${CLUSTER}.env"
mkdir -p "${remote_sshd_state_directory}"
chmod 700 "${remote_sshd_state_directory}"
remote_sshd_state_temporary_file="$(mktemp "${remote_sshd_state_file}.XXXXXX")"
{
printf 'CLUSTER=%s\n' "$CLUSTER"
printf 'REMOTE_USER=%s\n' "$REMOTE_USER"
printf 'NODE=%s\n' "$NODE"
printf 'PORT=%s\n' "$PORT"
} > "${remote_sshd_state_temporary_file}"
chmod 600 "${remote_sshd_state_temporary_file}"
mv "${remote_sshd_state_temporary_file}" "${remote_sshd_state_file}"
echo "Saved connection state: ${remote_sshd_state_file}"
echo "Connect with: $(cluster_shortcut "$CLUSTER") --service sshd"