{"id":14456,"date":"2026-07-10T16:10:48","date_gmt":"2026-07-10T20:10:48","guid":{"rendered":"https:\/\/labs.icahn.mssm.edu\/minervalab\/?page_id=14456"},"modified":"2026-07-10T16:12:56","modified_gmt":"2026-07-10T20:12:56","slug":"lsf-job-scheduler","status":"publish","type":"page","link":"https:\/\/labs.icahn.mssm.edu\/minervalab\/documentation\/lsf-job-scheduler\/","title":{"rendered":"LSF Job Scheduler"},"content":{"rendered":"<p>[et_pb_section fb_built=&#8221;1&#8243; fullwidth=&#8221;on&#8221; _builder_version=&#8221;4.16&#8243; _module_preset=&#8221;default&#8221; global_colors_info=&#8221;{}&#8221; theme_builder_area=&#8221;post_content&#8221;][et_pb_fullwidth_menu menu_id=&#8221;15&#8243; menu_style=&#8221;centered&#8221; fullwidth_menu=&#8221;on&#8221; active_link_color=&#8221;#d80b8c&#8221; dropdown_menu_bg_color=&#8221;#221f72&#8243; dropdown_menu_line_color=&#8221;#221f72&#8243; _builder_version=&#8221;4.16&#8243; _module_preset=&#8221;default&#8221; menu_font=&#8221;|600|||||||&#8221; menu_text_color=&#8221;#FFFFFF&#8221; menu_font_size=&#8221;16px&#8221; background_color=&#8221;#221f72&#8243; background_layout=&#8221;dark&#8221; global_colors_info=&#8221;{}&#8221; theme_builder_area=&#8221;post_content&#8221;][\/et_pb_fullwidth_menu][\/et_pb_section][et_pb_section fb_built=&#8221;1&#8243; _builder_version=&#8221;4.16&#8243; _module_preset=&#8221;default&#8221; custom_padding=&#8221;0px||0px||false|false&#8221; global_colors_info=&#8221;{}&#8221; theme_builder_area=&#8221;post_content&#8221;][et_pb_row _builder_version=&#8221;4.16&#8243; _module_preset=&#8221;default&#8221; custom_padding=&#8221;||0px||false|false&#8221; global_colors_info=&#8221;{}&#8221; theme_builder_area=&#8221;post_content&#8221;][et_pb_column type=&#8221;4_4&#8243; _builder_version=&#8221;4.16&#8243; _module_preset=&#8221;default&#8221; global_colors_info=&#8221;{}&#8221; theme_builder_area=&#8221;post_content&#8221;][et_pb_text admin_label=&#8221;Breadcrumb&#8221; _builder_version=&#8221;4.16&#8243; _module_preset=&#8221;default&#8221; global_colors_info=&#8221;{}&#8221; theme_builder_area=&#8221;post_content&#8221;]<a href=\"https:\/\/labs.icahn.mssm.edu\/minervalab\/scientific-computing-and-data\/\">Scientific Computing and Data<\/a>\u00a0\/\u00a0<a href=\"https:\/\/labs.icahn.mssm.edu\/minervalab\/\">High Performance Computing<\/a> \/ <a href=\"https:\/\/labs.icahn.mssm.edu\/minervalab\/documentation\/\">Documentation<\/a> \/ Load Sharing Facility (LSF) Job Scheduler[\/et_pb_text][\/et_pb_column][\/et_pb_row][\/et_pb_section][et_pb_section fb_built=&#8221;1&#8243; _builder_version=&#8221;4.16&#8243; global_colors_info=&#8221;{}&#8221; theme_builder_area=&#8221;post_content&#8221;][et_pb_row _builder_version=&#8221;4.16&#8243; background_size=&#8221;initial&#8221; background_position=&#8221;top_left&#8221; background_repeat=&#8221;repeat&#8221; global_colors_info=&#8221;{}&#8221; theme_builder_area=&#8221;post_content&#8221;][et_pb_column type=&#8221;4_4&#8243; _builder_version=&#8221;4.16&#8243; custom_padding=&#8221;|||&#8221; global_colors_info=&#8221;{}&#8221; custom_padding__hover=&#8221;|||&#8221; theme_builder_area=&#8221;post_content&#8221;][et_pb_text _builder_version=&#8221;4.27.4&#8243; header_font=&#8221;|600|||||||&#8221; header_text_color=&#8221;#221f72&#8243; header_font_size=&#8221;26px&#8221; header_2_text_color=&#8221;#221f72&#8243; header_2_font_size=&#8221;24px&#8221; background_size=&#8221;initial&#8221; background_position=&#8221;top_left&#8221; background_repeat=&#8221;repeat&#8221; hover_enabled=&#8221;0&#8243; global_colors_info=&#8221;{}&#8221; theme_builder_area=&#8221;post_content&#8221; custom_css_free_form=&#8221;.lsf-doc {||  &#8211;lsf-navy: #221f72;||  &#8211;lsf-blue: #00aeef;||  &#8211;lsf-red: #c9211e;||  &#8211;lsf-green: #009933;||  &#8211;lsf-stripe: #f5f5f5;||  &#8211;lsf-border: #dcdde3;||  &#8211;lsf-text: #26272b;||  &#8211;lsf-code-bg: #f4f5f7;||||  font-family: -apple-system, BlinkMacSystemFont, %22Segoe UI%22, Helvetica, Arial, sans-serif;||  color: var(&#8211;lsf-text);||  line-height: 1.6;||  max-width: 960px;||  margin: 0 auto;||}||||.lsf-doc * {||  box-sizing: border-box;||}||||.lsf-doc h1 {||  color: var(&#8211;lsf-navy);||  font-size: 1.9rem;||  margin: 0 0 16px;||}||||.lsf-doc h2 {||  color: var(&#8211;lsf-navy);||  font-size: 1.35rem;||  border-bottom: 2px solid var(&#8211;lsf-blue);||  padding-bottom: 6px;||  margin: 40px 0 16px;||}||||.lsf-doc h3 {||  color: var(&#8211;lsf-navy);||  font-size: 1.1rem;||  margin: 24px 0 10px;||}||||.lsf-doc p {||  margin: 0 0 14px;||}||||.lsf-doc a {||  color: var(&#8211;lsf-blue);||  text-decoration: none;||}||||.lsf-doc a:hover {||  text-decoration: underline;||}||||.lsf-doc code {||  background: var(&#8211;lsf-code-bg);||  border: 1px solid var(&#8211;lsf-border);||  border-radius: 3px;||  padding: 1px 5px;||  font-family: %22SFMono-Regular%22, Consolas, %22Liberation Mono%22, Menlo, monospace;||  font-size: 0.92em;||}||||.lsf-doc pre {||  background: var(&#8211;lsf-code-bg);||  border: 1px solid var(&#8211;lsf-border);||  border-left: 4px solid var(&#8211;lsf-blue);||  border-radius: 4px;||  padding: 14px 16px;||  overflow-x: auto;||  margin: 0 0 16px;||}||||.lsf-doc pre code {||  background: none;||  border: none;||  padding: 0;||  font-size: 0.88em;||  line-height: 1.55;||}||||.lsf-doc ul,||.lsf-doc ol {||  margin: 0 0 14px;||  padding-left: 26px;||}||||.lsf-doc li {||  margin-bottom: 6px;||}||||.lsf-doc .callout {||  color: var(&#8211;lsf-red);||  font-weight: 600;||}||||.lsf-doc .ok {||  color: var(&#8211;lsf-green);||  font-weight: 600;||}||||\/* &#8212;&#8212;&#8212;- Minerva-Res note &#8212;&#8212;&#8212;- *\/||.lsf-doc .cluster-note {||  background: #eef7fc;||  border: 1px solid #bfe4f5;||  border-left: 4px solid var(&#8211;lsf-blue);||  border-radius: 4px;||  padding: 10px 14px;||  margin: -6px 0 24px;||  font-size: 0.95rem;||}||||.lsf-doc .cluster-note strong {||  color: var(&#8211;lsf-navy);||}||||\/* &#8212;&#8212;&#8212;- Table of contents &#8212;&#8212;&#8212;- *\/||.lsf-doc .toc {||  background: var(&#8211;lsf-stripe);||  border: 1px solid var(&#8211;lsf-border);||  border-radius: 6px;||  padding: 16px 20px;||  margin: 20px 0 32px;||}||||.lsf-doc .toc-title {||  font-weight: 700;||  color: var(&#8211;lsf-navy);||  margin-bottom: 8px;||  display: block;||}||||.lsf-doc .toc ul {||  columns: 2;||  -webkit-columns: 2;||  list-style: none;||  padding-left: 0;||  margin: 0;||}||||.lsf-doc .toc li {||  margin-bottom: 6px;||}||||\/* &#8212;&#8212;&#8212;- Tables &#8212;&#8212;&#8212;- *\/||.lsf-doc table {||  width: 100%;||  border-collapse: collapse;||  margin: 0 0 16px;||  font-size: 0.93rem;||}||||.lsf-doc th {||  background: var(&#8211;lsf-blue);||  color: #fff;||  text-align: left;||  padding: 10px 12px;||}||||.lsf-doc td {||  padding: 10px 12px;||  border: 1px solid var(&#8211;lsf-border);||  vertical-align: top;||}||||.lsf-doc tr:nth-child(even) td {||  background: var(&#8211;lsf-stripe);||}||||.lsf-doc .queue-name {||  color: var(&#8211;lsf-navy);||  font-weight: 700;||}||||\/* Availability columns in the combined queue table *\/||.lsf-doc .queue-table th:nth-child(4),||.lsf-doc .queue-table th:nth-child(5),||.lsf-doc .queue-table td:nth-child(4),||.lsf-doc .queue-table td:nth-child(5) {||  text-align: center;||  width: 90px;||}||||.lsf-doc .queue-table td.yes {||  color: var(&#8211;lsf-green);||  font-weight: 700;||}||||.lsf-doc .queue-table td.no {||  color: #9a9ba3;||}||||\/* &#8212;&#8212;&#8212;- Responsive &#8212;&#8212;&#8212;- *\/||@media (max-width: 640px) {||  .lsf-doc .toc ul {||    columns: 1;||  }||||  .lsf-doc table,||  .lsf-doc thead,||  .lsf-doc tbody,||  .lsf-doc th,||  .lsf-doc td,||  .lsf-doc tr {||    display: block;||  }||||  .lsf-doc .queue-table thead {||    display: none;||  }||||  .lsf-doc .queue-table td {||    border: none;||    border-bottom: 1px solid var(&#8211;lsf-border);||  }||||  .lsf-doc .queue-table td.yes::before {||    content: %22Minerva: %22;||    color: var(&#8211;lsf-text);||    font-weight: 400;||  }||||  .lsf-doc .queue-table td.yes ~ td.yes::before,||  .lsf-doc .queue-table td:nth-child(5)::before {||    content: %22Minerva-Res: %22;||  }||}&#8221; sticky_enabled=&#8221;0&#8243;]<\/p>\n<div class=\"lsf-doc\">\n<h1>Load Sharing Facility (LSF) Job Scheduler<\/h1>\n<p>This guide covers job submission on both <strong>Minerva<\/strong> and <strong>Minerva-Res<\/strong>. The two systems share the same login process, the same <code>bsub<\/code> command and options, and the same sample job scripts below. The one difference is queue availability: <strong>Minerva-Res offers only the <code>express<\/code> and <code>premium<\/code> queues<\/strong>, while Minerva offers the full set listed in the <a href=\"#lsf9\">queue table<\/a>.<\/p>\n<p>When you log into Minerva (or Minerva-Res), you are placed on one of the login nodes. Login nodes should only be used for basic tasks such as file editing, code compilation, data backup, and job submission. <strong>The login nodes should not be used to run production jobs.<\/strong> Production work should be performed on the system&#8217;s compute nodes. Note that compute nodes do not have internet access by design for security reasons. You need to be on a login node or one of the interactive nodes to talk to the outside world.<\/p>\n<nav class=\"toc\"><span class=\"toc-title\">Contents<\/span><\/p>\n<ul>\n<li><a href=\"#lsf1\">Running Jobs on Compute Nodes<\/a><\/li>\n<li><a href=\"#lsf2\">Submitting Batch Jobs with bsub<\/a><\/li>\n<li><a href=\"#lsf3\">Commonly Used bsub Options<\/a><\/li>\n<li><a href=\"#lsf4\">Sample Batch Jobs<\/a><\/li>\n<li><a href=\"#lsf5\">Array Jobs<\/a><\/li>\n<li><a href=\"#lsf6\">Interactive Jobs<\/a><\/li>\n<li><a href=\"#lsf7\">Useful LSF Commands<\/a><\/li>\n<li><a href=\"#lsf8\">Pending Reasons<\/a><\/li>\n<li><a href=\"#lsf9\">LSF Queues And Policies<\/a><\/li>\n<li><a href=\"#lsf11\">Minerva Training Session<\/a><\/li>\n<\/ul>\n<\/nav>\n<h2 id=\"lsf1\">Running Jobs on Compute Nodes<\/h2>\n<p>Access to compute resources and job scheduling are managed by IBM Spectrum LSF (Load Sharing Facility) batch system. LSF is responsible for allocating resources to users, providing a framework for starting, executing and monitoring work on allocated resources, and scheduling work for future execution. Once access to compute resources has been allocated through the batch system, users have the ability to execute jobs on the allocated resources.<\/p>\n<p>You should request compute resources (e.g., number of nodes, number of compute cores per node, amount of memory, max time per job, etc.) that are consistent with the type of application(s) you are running:<\/p>\n<ul>\n<li>A serial (non-parallel) application can only make use of a single compute core on a single node, and will only see that node&#8217;s memory.<\/li>\n<li>A threaded program (e.g. one that uses OpenMP) employs a shared memory programming (SMP) model and is also restricted to a single node, but can run on multiple CPU cores on that same node. If multiple nodes are requested for a threaded application it will only see the memory and compute cores on the first node assigned to the job while those on the rest of the nodes are wasted. It&#8217;s highly recommended the number of threads is set to that of compute cores.<\/li>\n<li>An MPI (Message Passing Interface) parallel program can be distributed over multiple nodes: it launches multiple copies of its executable (MPI tasks, each assigned unique IDs called ranks) that can communicate with each other across the network.<\/li>\n<li>An LSF job array enables large numbers of independent jobs to be distributed over more than a single node from a single batch job submission.<\/li>\n<\/ul>\n<h2 id=\"lsf2\">Submitting Batch Jobs with <code>bsub<\/code><\/h2>\n<p>To submit a job to the LSF queue you must have a project allocation account first and it has to be provided using the <code>-P<\/code> option flag. To see the list of project allocation accounts you have access to:<\/p>\n<pre><code>$ mybalance<\/code><\/pre>\n<p>If you need access to a specific project account, you will have to have the project authorizer (owner\/delegate) send us a request at <a href=\"mailto:hpchelp@hpc.mssm.edu\">hpchelp@hpc.mssm.edu<\/a>.<\/p>\n<p>A batch job is the most common way users run production applications. To submit a batch job to one of the queues use the <strong>bsub<\/strong> command:<\/p>\n<p><strong>bsub [options] command<\/strong><\/p>\n<pre><code>$ bsub -P acc_hpcstaff -q premium -n 1 -W 00:10 -o hello.out echo \"Hello World!\"<\/code><\/pre>\n<p>In this example, a job is submitted to the premium queue (<code>-q premium<\/code>) to execute the command<\/p>\n<pre><code>echo \"Hello World!\"<\/code><\/pre>\n<p>with a resource request of a single core (<code>-n 1<\/code>) and the default amount of memory (3 GB) for a 10 min. of walltime (<code>-W 00:10<\/code>) under a specific project allocation account that Minerva staff members have access to (<code>-P acc_hpcstaff<\/code>). The result will be written to a file (<code>-o hello.out<\/code>) with a summary of resource usage information when the job is completed. The job will advance in the queue until it has reached the top. At this point, LSF will allocate the requested compute resources to the batch job.<\/p>\n<p>Typically, the user submits a job script to the batch system.<\/p>\n<p><strong>bsub [options] &lt; <em>YourJobScript<\/em><\/strong><\/p>\n<p>Here &#8220;<em>YourJobScript<\/em>&#8221; is the name of a text file containing <code>#BSUB<\/code> directives and shell commands that describe the particulars of the job you are submitting.<\/p>\n<pre><code>$ bsub &lt; HelloWorld.lsf<\/code><\/pre>\n<p>where HelloWorld.lsf is:<\/p>\n<pre><code>#!\/bin\/bash\r\n#BSUB -P acc_hpcstaff\r\n#BSUB -q premium\r\n#BSUB -n 1\r\n#BSUB -W 00:10\r\n#BSUB -e \"hello.err\"\r\n#BSUB -o \"hello.out\"\r\n\r\necho \"Hello World!\"<\/code><\/pre>\n<p>Note that if an option is given on both the bsub command line and in the job script, the command line option overrides the option in the script. The following job will be submitted to the express queue, not to the premium queue as specified in the job script.<\/p>\n<pre><code>$ bsub -q express &lt; HelloWorld.lsf<\/code><\/pre>\n<h2 id=\"lsf3\">Commonly Used bsub Options<\/h2>\n<table>\n<tbody>\n<tr>\n<td><code>-P<\/code> <em>acc_ProjectName<\/em><\/td>\n<td>Project allocation account name (required)<\/td>\n<\/tr>\n<tr>\n<td><code>-q<\/code> <em>queue_name<\/em><\/td>\n<td>Job submission queue (default: premium)<\/td>\n<\/tr>\n<tr>\n<td><code>-W<\/code> <em>WallClockTime<\/em><\/td>\n<td>Wall-clock limit in the form <em>HH:MM<\/em> (default 1:00)<\/td>\n<\/tr>\n<tr>\n<td><code>-J<\/code> <em>job_name<\/em><\/td>\n<td>Job name<\/td>\n<\/tr>\n<tr>\n<td><code>-n<\/code> <em>Ncore<\/em><\/td>\n<td>Total number of CPU cores requested (default: 1)<\/td>\n<\/tr>\n<tr>\n<td><code>-R rusage[mem=#]<\/code><\/td>\n<td>Amount of memory <strong>per core<\/strong> in MB (default: <code>rusage[mem=3000]<\/code>). Note that this is <em>not per job<\/em>.<\/p>\n<ul>\n<li>Max memory per node: ~1.4 TB (Chimera, himem, CATS, GPU H100, L40S), ~325 GB (GPU V100, A100), ~1.9 TB (himem-GPU A100-80GB), ~435 GB (GPU H100-80GB)<\/li>\n<\/ul>\n<\/td>\n<\/tr>\n<tr>\n<td><code>-R span[ptile=#n's per node]<\/code><\/td>\n<td>Number of CPU cores per physical node<\/td>\n<\/tr>\n<tr>\n<td><code>-R span[hosts=1]<\/code><\/td>\n<td>All cores on the same node<\/td>\n<\/tr>\n<tr>\n<td><code>-o<\/code> <em>output_file<\/em><\/td>\n<td>Direct job standard output to <em>output_file<\/em> (without <code>-e<\/code>, error also goes to this file). LSF appends output to the specified file; use <code>-oo<\/code> to overwrite instead.<\/td>\n<\/tr>\n<tr>\n<td><code>-e<\/code> <em>error_file<\/em><\/td>\n<td>Direct job error output to <em>error_file<\/em>. Use <code>-eo<\/code> to overwrite instead.<\/td>\n<\/tr>\n<tr>\n<td><code>-L<\/code> <em>login_shell<\/em><\/td>\n<td>Initializes the execution environment using the specified login shell<\/td>\n<\/tr>\n<\/tbody>\n<\/table>\n<p>For more details about bsub options, <a href=\"https:\/\/www.ibm.com\/docs\/en\/spectrum-lsf\/10.1.0?topic=bsub-options\" target=\"_blank\" rel=\"noopener\">click here<\/a>.<\/p>\n<p>Although there are default values for all batch parameters except <code>-P<\/code>, it is always a good idea to specify the name of the queue, the number of cores, and the walltime for all batch jobs. To minimize time spent waiting in the queue, specify the smallest walltime that will safely allow the job to complete.<\/p>\n<p>LSF inserts job report information into the job&#8217;s output. This includes the submitting user and host, the execution host, the CPU time (user plus system time) used by the job, and the exit status. Note that the standard LSF configuration allows emailing job output to the user on completion, <strong>but this feature has been disabled for the reason of stability.<\/strong><\/p>\n<h2 id=\"lsf4\">Sample Batch Jobs<\/h2>\n<h3>Serial Jobs<\/h3>\n<p>A serial job is one that only requires a single computational core. There is no queue specifically configured to run serial jobs. Serial jobs share nodes, rather than having exclusive access. Multiple jobs will be scheduled on an available node until either all cores are in use, or until there is not enough memory available for additional processes on that node. The following job script requests a single core and 8GB of memory for 2 hours in the &#8220;express&#8221; queue:<\/p>\n<pre><code>#!\/bin\/bash\r\n#BSUB -J mySerialJob                   # Job name\r\n#BSUB -P acc_YourAllocationAccount     # allocation account\r\n#BSUB -q express                       # queue\r\n#BSUB -n 1                             # number of compute cores = 1\r\n#BSUB -R rusage[mem=8000]              # 8GB of memory\r\n#BSUB -W 02:00                         # walltime in HH:MM\r\n#BSUB -o %J.stdout                     # output log (%J : JobID)\r\n#BSUB -eo %J.stderr                    # error log\r\n#BSUB -L \/bin\/bash                     # Initialize the execution environment\r\n\r\nml gcc\r\ncd \/sc\/arion\/work\/MyID\/my\/job\/dir\/\r\n..\/mybin\/serial_executable &lt; testdata.inp &gt; results.log<\/code><\/pre>\n<h3>Multithreaded Jobs<\/h3>\n<p>In general, a multithreaded application uses a single process which then spawns multiple threads of execution. The compute cores must be on the same node and it&#8217;s highly recommended the number of threads is set to the number of compute cores. The following example requests 8 cores on the same node for 12 hours in the &#8220;premium&#8221; queue. Note that the memory requirement (<code>-R rusage[mem=4000]<\/code>) is in MB and is PER CORE, not per job. A total of 32GB of memory will be allocated for this job.<\/p>\n<pre><code>#!\/bin\/bash\r\n#BSUB -J mySTARjob                # Job name\r\n#BSUB -P acc_PLK2                 # allocation account\r\n#BSUB -q premium                  # queue\r\n#BSUB -n 8                        # number of compute cores\r\n#BSUB -W 12:00                    # walltime in HH:MM\r\n#BSUB -R rusage[mem=4000]         # 32 GB of memory (4 GB per core)\r\n#BSUB -R span[hosts=1]            # all cores from the same node\r\n#BSUB -o %J.stdout                # output log (%J : JobID)\r\n#BSUB -eo %J.stderr               # error log\r\n#BSUB -L \/bin\/bash                # Initialize the execution environment\r\n\r\nmodule load star                  # load star module\r\nWRKDIR=\/sc\/arion\/projects\/hpcstaff\/benchmark_star\r\n\r\nSTAR --genomeDir $WRKDIR\/star-genome --readFilesIn Experiment1.fastq --runThreadN 8 -outFileNamePrefix Experiment1Star<\/code><\/pre>\n<h3>MPI Parallel Jobs<\/h3>\n<p>An MPI application launches multiple copies of its executable (MPI tasks) over multiple nodes that are separate but can communicate with each other across the network. The following example requests 48 cores and 2 hours in the &#8220;premium&#8221; queue. Those 48 cores are distributed over 6 nodes (8 cores per node) and a total of 192GB of memory will be allocated for this job, with 32GB on each node.<\/p>\n<pre><code>#!\/bin\/bash\r\n#BSUB -J myMPIjob                 # Job name\r\n#BSUB -P acc_hpcstaff             # allocation account\r\n#BSUB -q premium                  # queue\r\n#BSUB -n 48                       # total number of compute cores\r\n#BSUB -R span[ptile=8]            # 8 cores per node\r\n#BSUB -R rusage[mem=4000]         # 192 GB of memory (4 GB per core)\r\n#BSUB -W 02:00                    # walltime in HH:MM\r\n#BSUB -o %J.stdout\r\n#BSUB -eo %J.stderr\r\n#BSUB -L \/bin\/bash\r\n\r\nmodule load selfsched\r\nmpirun -np 48 selfsched &lt; test.inp<\/code><\/pre>\n<p>Please refer to <a href=\"https:\/\/labs.icahn.mssm.edu\/minervalab\/documentation\/gpgpu\/\" target=\"_blank\" rel=\"noopener\">the GPU section<\/a> for further information about options relevant to GPU job submissions.<\/p>\n<h2 id=\"lsf5\">Array Jobs<\/h2>\n<p>Sometimes it is necessary to run a group of jobs that share the same computational requirements but with different input files. Job arrays can be used to handle this type of embarrassingly parallel workload. They can be submitted, controlled, and monitored as a single unit or as individual jobs or groups of jobs. Each job submitted from a job array shares the same job ID as the job array and is uniquely referenced using an array index.<\/p>\n<p>To create a job array, add an index range to the jobname specification:<\/p>\n<pre><code>#BSUB -J MyArrayJob[1-100]<\/code><\/pre>\n<p>In this way you ask for 100 jobs, numbered from 1 to 100. Each job in the job array can be identified by its index, which is accessible through the environment variable <code>LSB_JOBINDEX<\/code>. LSF provides the runtime variables <code>%I<\/code> and <code>%J<\/code>, which correspond to the job array index and the jobID respectively. They can be used in the <code>#BSUB<\/code> option specification to diversify the jobs. In your commands, however, you must use the environment variables <code>LSB_JOBINDEX<\/code> and <code>LSB_JOBID<\/code>.<\/p>\n<pre><code>#!\/bin\/bash\r\n#BSUB -P acc_hpcstaff\r\n#BSUB -n 1\r\n#BSUB -W 02:00\r\n#BSUB -q express\r\n#BSUB -J \"jobarraytest[1-10]\"\r\n#BSUB -o out.%J.%I\r\n#BSUB -e err.%J.%I\r\n\r\necho \"Working on file.$LSB_JOBINDEX\"<\/code><\/pre>\n<p>A total of 10 jobs, jobarraytest[1] to jobarraytest[10], will be created and submitted to the queue simultaneously by this script. The output and error files will be written to <code>out.JobID.1<\/code> ~ <code>out.JobID.10<\/code> and <code>err.JobID.1<\/code> ~ <code>err.JobID.10<\/code>, respectively.<\/p>\n<h2 id=\"lsf6\">Interactive Jobs<\/h2>\n<p>Interactive batch jobs give users interactive access to compute resources. A common use for interactive batch jobs is debugging and testing. Running a batch-interactive job is done by using the <code>-I<\/code> option with <strong>bsub<\/strong>.<\/p>\n<p>Here is an example command creating an interactive shell on compute nodes <em>with internet access:<\/em><\/p>\n<pre><code>bsub -P AllocationAccount -q interactive -n 8 -W 15 -R span[hosts=1] -XF -Is \/bin\/bash<\/code><\/pre>\n<p>This command allocates a total of 8 cores on one of the interactive compute nodes, reserved exclusively for jobs running in interactive mode. All cores are on the same node. The <code>-XF<\/code> option enables X11 forwarding so graphics applications can open on your screen. Once the interactive shell has started, the user can execute jobs interactively there. In addition to those dedicated nodes, regular compute nodes with no internet access can also be used to open an interactive session by submitting a batch interactive job to non-interactive queues such as &#8220;premium&#8221;, &#8220;express&#8221;, and &#8220;gpu&#8221; using the <code>bsub -I<\/code>, <code>-Is<\/code>, and <code>-Ip<\/code> options. On these nodes, internet access can be enabled after issuing the command <code>module load proxies<\/code>.<\/p>\n<h2 id=\"lsf7\">Useful LSF Commands<\/h2>\n<p><strong><a href=\"https:\/\/www.ibm.com\/docs\/en\/spectrum-lsf\/10.1.0?topic=reference-bjobs\" target=\"_blank\" rel=\"noopener\">bjobs<\/a><\/strong> \u2014 Show job status. Check the status of your own jobs in the queue. To display detailed information about a job in a multi-line format use <code>bjobs -l JobID<\/code>. If a job has already completed, <code>bjobs<\/code> won&#8217;t show information about it \u2014 use <a href=\"https:\/\/www.ibm.com\/docs\/en\/spectrum-lsf\/10.1.0?topic=reference-bhist\" target=\"_blank\" rel=\"noopener\">bhist<\/a> to retrieve the job&#8217;s record from the LSF database.<\/p>\n<p><strong><a href=\"https:\/\/www.ibm.com\/docs\/en\/spectrum-lsf\/10.1.0?topic=reference-bkill\" target=\"_blank\" rel=\"noopener\">bkill<\/a><\/strong> \u2014 Cancel a batch job. A job can be removed from the queue or killed if running, using <code>bkill JobID<\/code>. There are many ways to terminate a specific set of jobs:<\/p>\n<ul>\n<li>To terminate a job by its job name: <code>bkill -J myjob_1<\/code><\/li>\n<li>To terminate a bunch of jobs using a wildcard: <code>bkill -J myjob_*<\/code><\/li>\n<li>To kill some selected array jobs: <code>bkill JobID[1,7,10-25]<\/code><\/li>\n<li>To kill all your jobs: <code>bkill 0<\/code><\/li>\n<\/ul>\n<p><strong><a href=\"https:\/\/www.ibm.com\/docs\/en\/spectrum-lsf\/10.1.0?topic=reference-bmod\" target=\"_blank\" rel=\"noopener\">bmod<\/a><\/strong> \u2014 Modify the resource requirement of a pending job. Many batch job specifications can be modified after submission and <strong>before it runs<\/strong> \u2014 typically job size (number of nodes), queue, and wall clock limit. Job specifications cannot be modified once a job enters the <code>RUN<\/code> state. For example: <code>bmod -q express jobID<\/code> changes the job&#8217;s queue to express; <code>bmod -R rusage[mem=20000] jobID<\/code> changes the job&#8217;s memory requirement to 20 GB per core. Note that <code>-R<\/code> replaces <strong>ALL<\/strong> R fields, not just the one you specify.<\/p>\n<p><strong><a href=\"https:\/\/www.ibm.com\/docs\/en\/spectrum-lsf\/10.1.0?topic=reference-bpeek\" target=\"_blank\" rel=\"noopener\">bpeek<\/a><\/strong> \u2014 Displays the stdout and stderr output of an unfinished job.<\/p>\n<p><strong><a href=\"https:\/\/www.ibm.com\/docs\/en\/spectrum-lsf\/10.1.0?topic=reference-bqueues\" target=\"_blank\" rel=\"noopener\">bqueues<\/a><\/strong> \u2014 Displays information about queues.<\/p>\n<p><strong><a href=\"https:\/\/www.ibm.com\/docs\/en\/spectrum-lsf\/10.1.0?topic=reference-bhosts\" target=\"_blank\" rel=\"noopener\">bhosts<\/a><\/strong> \u2014 Displays hosts and their static and dynamic resources.<\/p>\n<p>For the full list of LSF commands, see the <a href=\"https:\/\/www.ibm.com\/docs\/en\/spectrum-lsf\/10.1.0?topic=reference-command\" target=\"_blank\" rel=\"noopener\">IBM Spectrum LSF Command Reference<\/a>.<\/p>\n<h2 id=\"lsf8\">Pending Reasons<\/h2>\n<p>There could be many reasons for a job to stay in the queue longer than usual: the whole cluster may be busy; your job may request a large amount of compute resources (memory, cores) that no compute node can currently satisfy; or your job may overlap with a scheduled PM. To see the pending reason, use the bjobs command with the <code>-l<\/code> flag:<\/p>\n<pre><code>$ bjobs -l<\/code><\/pre>\n<h2 id=\"lsf9\">LSF Queues And Policies<\/h2>\n<p>The command to check the queues is <strong>bqueues<\/strong>. To get more details about a specific queue, type <strong>bqueues -l<\/strong>, e.g. <strong>bqueues -l premium<\/strong>.<\/p>\n<p>* The default memory for all queues is 3000 MB.<\/p>\n<table class=\"queue-table\">\n<thead>\n<tr>\n<th>Queue<\/th>\n<th>Description<\/th>\n<th>Max Walltime<\/th>\n<th>Minerva<\/th>\n<th>Minerva-Res<\/th>\n<\/tr>\n<\/thead>\n<tbody>\n<tr>\n<td class=\"queue-name\">premium<\/td>\n<td>Normal submission queue<\/td>\n<td>144 hrs<\/td>\n<td class=\"yes\">\u2713<\/td>\n<td class=\"yes\">\u2713<\/td>\n<\/tr>\n<tr>\n<td class=\"queue-name\">express<\/td>\n<td>Rapid turnaround jobs<\/td>\n<td>12 hrs<\/td>\n<td class=\"yes\">\u2713<\/td>\n<td class=\"yes\">\u2713<\/td>\n<\/tr>\n<tr>\n<td class=\"queue-name\">interactive<\/td>\n<td>Jobs running in interactive mode<\/td>\n<td>12 hrs<\/td>\n<td class=\"yes\">\u2713<\/td>\n<td class=\"no\">\u2014<\/td>\n<\/tr>\n<tr>\n<td class=\"queue-name\">long<\/td>\n<td>Jobs requiring extended runtime<\/td>\n<td>336 hrs<\/td>\n<td class=\"yes\">\u2713<\/td>\n<td class=\"no\">\u2014<\/td>\n<\/tr>\n<tr>\n<td class=\"queue-name\">gpu<\/td>\n<td>Jobs requiring GPU resources<\/td>\n<td>144 hrs<\/td>\n<td class=\"yes\">\u2713<\/td>\n<td class=\"no\">\u2014<\/td>\n<\/tr>\n<tr>\n<td class=\"queue-name\">gpuexpress<\/td>\n<td>Short jobs requiring GPU resources<\/td>\n<td>15 hrs<\/td>\n<td class=\"yes\">\u2713<\/td>\n<td class=\"no\">\u2014<\/td>\n<\/tr>\n<tr>\n<td class=\"queue-name\">ondemand<\/td>\n<td>Jobs dedicated to Open OnDemand<\/td>\n<td>12 hrs<\/td>\n<td class=\"yes\">\u2713<\/td>\n<td class=\"no\">\u2014<\/td>\n<\/tr>\n<tr>\n<td class=\"queue-name\">private<\/td>\n<td>Jobs using dedicated resources<\/td>\n<td>Unlimited<\/td>\n<td class=\"yes\">\u2713<\/td>\n<td class=\"no\">\u2014<\/td>\n<\/tr>\n<tr>\n<td class=\"queue-name\">others<\/td>\n<td>Any other queues are for testing by the Scientific Computing group<\/td>\n<td>N\/A<\/td>\n<td class=\"yes\">\u2713<\/td>\n<td class=\"no\">\u2014<\/td>\n<\/tr>\n<\/tbody>\n<\/table>\n<p class=\"cluster-note\"><strong>Minerva-Res note:<\/strong> only <code>express<\/code> and <code>premium<\/code> are available on Minerva-Res. All other queues in the table above are Minerva-only.<\/p>\n<h2>Policies<\/h2>\n<p>LSF is configured to do &#8220;absolute priority scheduling (APS)&#8221; backfill. Backfilling allows smaller, shorter jobs to use otherwise idle resources.<\/p>\n<p>To check the priority of pending jobs: <code>$ bjobs -u all -aps<\/code><\/p>\n<p>In certain special cases, the priority of a job may be manually increased upon request. To request a priority change, contact the Minerva HPC team at <a href=\"mailto:hpchelp@hpc.mssm.edu\">hpchelp@hpc.mssm.edu<\/a>. We will need the job ID and reason to submit the request.<\/p>\n<h2>Improper Resource Specification<\/h2>\n<p>Minerva is a shared resource. Improper specification of job requirements not only wastes them but also prevents other researchers from running their jobs.<\/p>\n<p><strong>1. If your program is not explicitly written to use more than one core, specifying more than one core wastes cores that could be used by other users.<\/strong> E.g.:<\/p>\n<p>IF <code>bsub -n 6 &lt; single_core_program<\/code> THEN:<\/p>\n<ul>\n<li><strong>6 cores allocated<\/strong><\/li>\n<li>program runs on 1 core<\/li>\n<li><span class=\"callout\">5 cores are idle<\/span><\/li>\n<\/ul>\n<p><strong>2. If your program can use more than one core but does not use tools like mpirun, mpiexec, or torchrun, then all cores must be on the same node.<\/strong> This is a shared-memory multiprocessing (SMP) program. You must specify <code>-R span[hosts=1]<\/code> to ensure all cores are on the same node.<\/p>\n<p>E.g.: IF <code>bsub -n 6 &lt; SMP_program<\/code> THEN:<\/p>\n<ul>\n<li><strong>6 cores are allocated: 1 on node 1, 2 on node 2, 3 on node 3<\/strong><\/li>\n<li>1 core on node 1 is used to run the program<\/li>\n<li><span class=\"callout\">The other 5 cores sit idle<\/span><\/li>\n<\/ul>\n<p>or, perhaps, IF <code>bsub -n 6 &lt; SMP_program<\/code> THEN:<\/p>\n<ul>\n<li><strong>6 cores are allocated: 3 on node 1, 1 on node 2, 2 on node 3<\/strong><\/li>\n<li>3 cores on node 1 are used to run the program<\/li>\n<li><span class=\"callout\">The other 3 cores sit idle<\/span><\/li>\n<\/ul>\n<p>But with <code>-R span[hosts=1]<\/code>: IF <code>bsub -n 6 -R span[hosts=1] &lt; SMP_program<\/code> THEN:<\/p>\n<ul>\n<li><strong>6 cores are allocated: 6 on node 1<\/strong><\/li>\n<li><span class=\"ok\">All 6 cores on node 1 are used to run the program<\/span><\/li>\n<\/ul>\n<p><strong>3. Memory specified by <code>-R rusage[mem=xxx]<\/code> is reserved for your use and cannot be used by anyone else until released at the end of the job.<\/strong> If most\/all of the memory on a node is reserved, no additional jobs can run even if CPUs are still available.<\/p>\n<p>Example \u2014 IF:<\/p>\n<ul>\n<li>A 192GB node has 3 jobs dispatched to it<\/li>\n<li>Each job requests 1 core and 64GB of memory (<code>-n 1 -R rusage[mem=64G]<\/code>)<\/li>\n<li>Each is actually using only 1GB<\/li>\n<\/ul>\n<p>THEN:<\/p>\n<ul>\n<li>All memory on the node is reserved, so no other job can be dispatched to it.<\/li>\n<li>Only 3GB of memory is being used.<\/li>\n<li><span class=\"callout\">189 GB of memory and 43 cores are idle and cannot be used.<\/span><\/li>\n<\/ul>\n<p><strong>4. Check to see how much memory your program is actually using and request accordingly.<\/strong> If you&#8217;re running a series of jobs using the same program, run one or two test jobs with a large amount of memory and then adjust for the production runs.<\/p>\n<p>Test:<\/p>\n<pre><code>bsub -R rusage[mem=6G] other options &lt; testJob.lsf<\/code><\/pre>\n<p>Output:<\/p>\n<pre><code>Successfully completed.\r\nResource usage summary:\r\n    CPU time :             1.82 sec.\r\n    Max Memory :           8 MB\r\n    Average Memory :       6.75 MB\r\n    Total Requested Memory : 6144.00 MB\r\n    Delta Memory :         6136.00 MB\r\n    Max Swap :             -\r\n    Max Processes :        3\r\n    Max Threads :          3\r\n    Run time :              2 sec.\r\n    Turnaround time :      50 sec.<\/code><\/pre>\n<p>The test run shows this program used 8MB maximum. Memory usage varies between runs depending on what else is running on the node, so scale the request down but leave some &#8220;wiggle room&#8221;:<\/p>\n<pre><code>bsub -R rusage[mem=15M] other options &lt; productionJob.lsf<\/code><\/pre>\n<pre><code>Successfully completed.\r\nResource usage summary:\r\n    CPU time :             2.93 sec.\r\n    Max Memory :           5 MB\r\n    Average Memory :       4.00 MB\r\n    Total Requested Memory : 15.00 MB\r\n    Delta Memory :         10.00 MB\r\n    Max Swap :             -\r\n    Max Processes :        3\r\n    Max Threads :          3\r\n    Run time :              4 sec.\r\n    Turnaround time :      47 sec.<\/code><\/pre>\n<h2 id=\"lsf11\">Minerva Training Session<\/h2>\n<p>Each spring and fall, the Scientific Computing and Data team hosts a series of training sessions, including one on the LSF Job Scheduler. <a href=\"https:\/\/labs.icahn.mssm.edu\/minervalab\/resources\/the-minerva-user-group-and-training-classes\/\" target=\"_blank\" rel=\"noopener\">Click here<\/a> to see past and future Minerva training sessions.<\/p>\n<\/div>\n<p>[\/et_pb_text][\/et_pb_column][\/et_pb_row][\/et_pb_section]<\/p>\n","protected":false},"excerpt":{"rendered":"<p>Scientific Computing and Data\u00a0\/\u00a0High Performance Computing \/ Documentation \/ Load Sharing Facility (LSF) Job Scheduler Load Sharing Facility (LSF) Job Scheduler This guide covers job submission on both Minerva and Minerva-Res. The two systems share the same login process, the same bsub command and options, and the same sample job scripts below. The one difference [&hellip;]<\/p>\n","protected":false},"author":624,"featured_media":0,"parent":35,"menu_order":0,"comment_status":"closed","ping_status":"closed","template":"","meta":{"_et_pb_use_builder":"on","_et_pb_old_content":"","_et_gb_content_width":"","footnotes":""},"class_list":["post-14456","page","type-page","status-publish","hentry"],"aioseo_notices":[],"_links":{"self":[{"href":"https:\/\/labs.icahn.mssm.edu\/minervalab\/wp-json\/wp\/v2\/pages\/14456","targetHints":{"allow":["GET"]}}],"collection":[{"href":"https:\/\/labs.icahn.mssm.edu\/minervalab\/wp-json\/wp\/v2\/pages"}],"about":[{"href":"https:\/\/labs.icahn.mssm.edu\/minervalab\/wp-json\/wp\/v2\/types\/page"}],"author":[{"embeddable":true,"href":"https:\/\/labs.icahn.mssm.edu\/minervalab\/wp-json\/wp\/v2\/users\/624"}],"replies":[{"embeddable":true,"href":"https:\/\/labs.icahn.mssm.edu\/minervalab\/wp-json\/wp\/v2\/comments?post=14456"}],"version-history":[{"count":4,"href":"https:\/\/labs.icahn.mssm.edu\/minervalab\/wp-json\/wp\/v2\/pages\/14456\/revisions"}],"predecessor-version":[{"id":14483,"href":"https:\/\/labs.icahn.mssm.edu\/minervalab\/wp-json\/wp\/v2\/pages\/14456\/revisions\/14483"}],"up":[{"embeddable":true,"href":"https:\/\/labs.icahn.mssm.edu\/minervalab\/wp-json\/wp\/v2\/pages\/35"}],"wp:attachment":[{"href":"https:\/\/labs.icahn.mssm.edu\/minervalab\/wp-json\/wp\/v2\/media?parent=14456"}],"curies":[{"name":"wp","href":"https:\/\/api.w.org\/{rel}","templated":true}]}}