-
Notifications
You must be signed in to change notification settings - Fork 2
Expand file tree
/
Copy pathstd-kernel.py
More file actions
361 lines (292 loc) · 11.7 KB
/
Copy pathstd-kernel.py
File metadata and controls
361 lines (292 loc) · 11.7 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
#!/usr/bin/python
# Copyright (C) 2018 Julien Peloton
#
# This program is free software: you can redistribute it and/or modify
# it under the terms of the GNU General Public License as published by
# the Free Software Foundation, either version 3 of the License, or
# (at your option) any later version.
#
# This program is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# GNU General Public License for more details.
#
# You should have received a copy of the GNU General Public License
# along with this program. If not, see <http://www.gnu.org/licenses/>.
"""
Generate Jupyter kernels to use Apache Spark at NERSC
Author: Julien Peloton, peloton@lal.in2p3.fr
"""
import os
import stat
import argparse
from kernel_util import safe_mkdir
def create_startup_file(path, spark_version):
"""
Create a startup file (bash) to load Spark module, and start a cluster.
Parameters
----------
path : str
Where to store the startup file
spark_version: str
Apache Spark version. Only non-shifter versions are supported, that is
version 2.0.0 and 2.1.0.
Returns
----------
filename: str
Returns the startup script filename (with full path)
"""
filename = os.path.join(path, 'start_spark.sh')
pythonpath = "/usr/common/software/python/3.5-anaconda/bin/python"
with open(filename, 'w') as f:
print('#!/bin/bash', file=f)
print('module load spark/{}'.format(spark_version), file=f)
print('start-all.sh', file=f)
print('{} -m ipykernel $@'.format(pythonpath), file=f)
# Change permission to rwx for the user
os.chmod(filename, stat.S_IRWXU)
return filename
def create_standard_kernel(
path, startupname, kernelname,
spark_version, pyspark_args):
"""
Create a JSON file with the kernel properties.
Suitable for Spark versions ran using modules (<= 2.1.0 - deprecated).
Parameters
----------
path : str
Where to store the kernel file
startupname : str
Startup script filename (with full path) which load Spark module,
and start the cluster. See `create_startup_file`.
kernelname : str
Name of the kernel (will be displayed on the UI).
spark_version: str
Apache Spark version. Only non-shifter versions are supported, that is
version 2.0.0 and 2.1.0.
pyspark_args: str
Extra arguments to pass to pyspark. Typically:
--master local[n] --packages <> --jars <>
See https://spark.apache.org/docs/latest/submitting-applications.html
for more information.
"""
software_path = "/global/common/cori/software"
spark_path = "{}/spark/{}".format(software_path, spark_version)
filename = os.path.join(path, 'kernel.json')
with open(filename, 'w') as f:
print('{', file=f)
# Displayed name of the cluster
print(' "display_name": "{} ({})",'.format(
kernelname, spark_version), file=f)
# Kernel language is Python
print(' "language": "python",', file=f)
# Startup args
print(' "argv": [', file=f)
# Startup script to launch the Spark cluster
print(' "{}",'.format(startupname), file=f)
# Other required args to start the kernel
print(' "-m",', file=f)
print(' "ipykernel",', file=f)
print(' "-f",', file=f)
print(' "{connection_file}"', file=f)
print(' ],', file=f)
# Environment
print(' "env": {', file=f)
# Spark installation path
print(' "SPARK_HOME": "{}",'.format(spark_path), file=f)
# Arguments to be passed to pyspark
print(' "PYSPARK_SUBMIT_ARGS": ', file=f)
# --> Resources
print(' "{} pyspark-shell",'.format(pyspark_args), file=f)
# Pyspark startup shell script
print(' "PYTHONSTARTUP": "{}/python/pyspark/shell.py",'.format(
spark_path), file=f)
# Need to include py4j library
print(' "PYTHONPATH":', file=f)
print(' "{}/python:{}/python/lib/py4j-0.10.4-src.zip",'.format(
spark_path, spark_path), file=f)
# Version of Python. Only work for 3.5 for the moment.
print(' "PYSPARK_PYTHON": ', file=f)
print('"/usr/common/software/python/3.5-anaconda/bin/python",', file=f)
# Ipython driver
print(' "PYSPARK_DRIVER_PYTHON": "ipython3"', file=f)
print(' }', file=f)
print('}', file=f)
def create_shifter_kernel(
path, kernelname, spark_version, pyspark_args, shifter_image=None):
"""
Create a JSON file with the kernel properties.
Suitable for Spark versions ran inside of shifter (2.3.0+).
Parameters
----------
path : str
Where to store the kernel file
kernelname : str
Name of the kernel (will be displayed on the UI).
spark_version: str
Apache Spark version. Only shifter versions are supported, that is
version 2.3.0+. If shifter_image is set, this option has no effect.
pyspark_args: str
Extra arguments to pass to pyspark. Typically:
--master local[n] --packages <> --jars <>
See https://spark.apache.org/docs/latest/submitting-applications.html
for more information.
shifter_image : None or str
If not None, the name of the user-defined shifter image to load.
Default is None. Bypass the spark_version option.
"""
# Software path inside of Shifter
software_path = "/usr/local/bin/"
spark_path = "{}/spark-{}".format(software_path, spark_version)
filename = os.path.join(path, 'kernel.json')
# Folder to store temporary files
if ("SCRATCH" in os.environ):
scratch = os.environ["SCRATCH"]
else:
scratch = path
tmpfolder = "{}/tmpfiles".format(scratch)
safe_mkdir(tmpfolder, True)
with open(filename, 'w') as f:
print('{', file=f)
# Displayed name of the cluster
print(' "display_name": "{} ({})",'.format(
kernelname, spark_version), file=f)
# Kernel language is Python
print(' "language": "python",', file=f)
# Startup args
print(' "argv": [', file=f)
# Run Spark inside of Shifter
print(' "shifter",', file=f)
if shifter_image:
print(' "--image={}",'.format(shifter_image), file=f)
else:
print(
' "--image=nersc/spark-{}:v1",'.format(spark_version),
file=f)
print(' "--volume=\\"{}:/tmp:perNodeCache=size=200G\\"",'.format(
tmpfolder), file=f)
print(' "/root/anaconda3/bin/python",', file=f)
# Other required args to start the kernel
print(' "-m",', file=f)
print(' "ipykernel",', file=f)
print(' "-f",', file=f)
print(' "{connection_file}"', file=f)
print(' ],', file=f)
# Environment
print(' "env": {', file=f)
# Spark installation path
print(' "SPARK_HOME": "{}",'.format(spark_path), file=f)
# Arguments to be passed to pyspark
print(' "PYSPARK_SUBMIT_ARGS": ', file=f)
# --> Resources
print(' "{} pyspark-shell",'.format(pyspark_args), file=f)
# Pyspark startup shell script
print(' "PYTHONSTARTUP": "{}/python/pyspark/shell.py",'.format(
spark_path), file=f)
# Need to include py4j library
print(' "PYTHONPATH":', file=f)
print(' "{}/python:{}/python/lib/py4j-0.10.6-src.zip",'.format(
spark_path, spark_path), file=f)
# Version of Python. Only work for 3.5 for the moment.
print(' "PYSPARK_PYTHON": "/root/anaconda3/bin/python",', file=f)
# Ipython driver
print(' "PYSPARK_DRIVER_PYTHON": "ipython3",', file=f)
print(' "JAVA_HOME": "/usr"', file=f)
print(' }', file=f)
print('}', file=f)
def addargs(parser):
""" Parse command line arguments for spark-kernel-nersc """
parser.add_argument(
'-kernelname', dest='kernelname',
required=True,
help='Name of the Jupyter kernel to be displayed')
parser.add_argument(
'-spark_version', dest='spark_version',
default="2.3.0",
help="""
Version of Apache Spark. Available: 2.0.0, 2.1.0, 2.3.0.
Note that 2.0.0, and 2.1.0 are standard kernels, while 2.3.0 makes use
of shifter to run. Default is 2.3.0.
This option has no effect if `-shifter_image` is provided.
""")
parser.add_argument(
'-shifter_image', dest='shifter_image',
default=None,
help="""
Custom shifter image with Spark plus additional dependencies.
See the README of this repo for more information. Default is None.
This option automatically bypasses `-spark_version`.
""")
parser.add_argument(
'-pyspark_args', dest='pyspark_args',
default="--master local[4]",
help="""
Submission arguments for pyspark.
See https://spark.apache.org/docs/latest/submitting-applications.html
for more information. Default is "--master local[4]".
""")
parser.add_argument(
'--local', dest='local',
action="store_true",
help="""
If specified, kernel and startup scripts will be dump on the current
working directory instead of the NERSC kernel folder. Default is False.
""")
if __name__ == "__main__":
""" Create Jupyter kernels for using Apache Spark at NERSC
Launch it using `python std-kernel.py <args>`.
Run `python std-kernel.py --help` for more information on inputs.
"""
parser = argparse.ArgumentParser(
description='Create Jupyter kernels for using Apache Spark at NERSC')
addargs(parser)
args = parser.parse_args(None)
msg = """
Note:
----------------
The large-memory login node used by https://jupyter-dev.nersc.gov/
is a shared resource, so please be careful not to use too many CPUs
or too much memory.
That means do not specify `--master local[*]` in your kernel, but limit
the resource to a few core. Typically `--master local[4]` is enough for
prototyping a program.
"""
print(msg)
# Grab $HOME path
HOME = os.environ['HOME']
if not args.local:
# Kernels are stored here
# See https://jupyter-client.readthedocs.io/en/stable/kernels.html#kernel-specs
path = '{}/.local/share/jupyter/kernels/{}'.format(HOME, args.kernelname)
else:
path = os.curdir
# Create the folder to store the kernel if needed,
# and store the kernel + the startup script
safe_mkdir(path, verbose=True)
valid = True
if args.shifter_image is not None:
print("Loading custom shifter image...")
# Create the kernel
create_shifter_kernel(
path, args.kernelname, args.spark_version,
args.pyspark_args, args.shifter_image)
elif args.spark_version <= "2.1.0":
# Startup file to load the Spark module
startup_fn = create_startup_file(path, args.spark_version)
# Create the kernel
create_standard_kernel(
path, startup_fn, args.kernelname,
args.spark_version, args.pyspark_args)
elif args.spark_version == "2.3.0":
# Create the kernel
create_shifter_kernel(
path, args.kernelname, args.spark_version,
args.pyspark_args, None)
else:
print("""
Kernel type not understood! Nothing has been created.
Run `python std-kernel.py --help` for more information on inputs.
""")
valid = False
if valid:
print("Kernel stored at {}".format(path))