From 5a0f98802cab73dfcdaef01b1200e96bc6b9a8ae Mon Sep 17 00:00:00 2001 From: Nicola Soranzo Date: Wed, 6 Jun 2018 12:22:24 +0100 Subject: [PATCH] Strip a spurious Slurm warning from job stderr Reason for the Slurm warning ---------------------------- The Linux kernel memory controller (responsible for the memory cgroup) may run out of cgroup subsystem state (CSS) IDs. When a cgroup is created it is assigned a CSS ID to manage its state. Upon removal of the cgroup the corresponding state information (e.g. cache entries) may still exist and therefore the CSS ID is still held. When multiple frequent short-lived jobs are run on a cluster node, the number of available CSS IDs becomes exhausted and creating a new memory cgroup results in a ENOSPC ("No space left on device"), which is reported by SLURM with the message "unable to add task[pid=] to memory cg '(null)'" (even if jobs run fine). There is a bugfix for the kernel that releases the CSS ID upon cgroup destruction, but it appears to be only available after Linux 4.4. We use CentOS 7, which is based on Linux 3.10. The only temporary solutions is to reboot the affected cluster node. Thanks to @tuxtobin for the detailed analysis above. Reason for this patch --------------------- Many tools rely on an empty stderr to determine if the job was successful, either because they were never updated to use ``/`detect_errors`, or because the underlying tool returns a non-zero exit code when successful, e.g. `tranalign` from https://toolshed.g2.bx.psu.edu/view/devteam/emboss_5/832c20329690 . Even using a `` inside `` would not work because the Slurm warning contains the word `error`. --- lib/galaxy/jobs/runners/slurm.py | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/lib/galaxy/jobs/runners/slurm.py b/lib/galaxy/jobs/runners/slurm.py index c4c38163f9a..5c7856f1dfc 100644 --- a/lib/galaxy/jobs/runners/slurm.py +++ b/lib/galaxy/jobs/runners/slurm.py @@ -3,7 +3,10 @@ SLURM job control via the DRMAA API. """ import logging import os +import re +import shutil import subprocess +import tempfile import time from galaxy import model @@ -25,6 +28,7 @@ SLURM_MEMORY_LIMIT_EXCEEDED_MSG = 'slurmstepd: error: Exceeded job memory limit' SLURM_MEMORY_LIMIT_EXCEEDED_PARTIAL_WARNINGS = [': Exceeded job memory limit at some point.', ': Exceeded step memory limit at some point.'] SLURM_MEMORY_LIMIT_SCAN_SIZE = 16 * 1024 * 1024 # 16MB +SLURM_UNABLE_TO_ADD_TASK_TO_MEMORY_CG_MSG_RE = re.compile(r"""slurmstepd: error: task/cgroup: unable to add task\[pid=\d+\] to memory cg '\(null\)'$""") # These messages are returned to the user OUT_OF_MEMORY_MSG = 'This job was terminated because it used more memory than it was allocated.' @@ -149,6 +153,14 @@ class SlurmJobRunner(DRMAAJobRunner): self.work_queue.put((self.fail_job, ajs)) return if drmaa_state == self.drmaa_job_states.DONE: + with open(ajs.error_file, 'r') as rfh: + first_line = rfh.readline() + if SLURM_UNABLE_TO_ADD_TASK_TO_MEMORY_CG_MSG_RE.match(first_line): + with tempfile.NamedTemporaryFile('w', delete=False) as wfh: + shutil.copyfileobj(rfh, wfh) + wf_name = wfh.name + shutil.move(wf_name, ajs.error_file) + log.debug('(%s/%s) Job completed, removing SLURM spurious warning: "%s"', ajs.job_wrapper.get_id_tag(), ajs.job_id, first_line) with open(ajs.error_file, 'r+') as f: if os.path.getsize(ajs.error_file) > SLURM_MEMORY_LIMIT_SCAN_SIZE: f.seek(-SLURM_MEMORY_LIMIT_SCAN_SIZE, os.SEEK_END)