From 3e034a49b36030611bc6f9c4dbfda4e553f8d163 Mon Sep 17 00:00:00 2001 From: Webserver Operations Date: Mon, 21 Sep 2026 20:18:19 -0600 Subject: [PATCH 1/2] fix: detach async jobs by forking, not by killing the web server An async submit has to finish an HTTP response now and keep executing the query afterwards. Under CGI those pull against each other: the server completes the response when the script closes stdout, and mod_cgi terminates the script once the request is cleaned up. So the process that runs the query can be neither the one holding stdout nor a member of the request's process group. __printAsyncResponse__ resolved that by sending the 303, sleeping two seconds, and then SIGKILLing os.getppid(). Under CGI that parent is the web server child serving the request. Killing it does end the response, since the body carried no Content-Length, and it does orphan the CGI so the job survives. But the client's connection dies mid-message, and behind a reverse proxy the proxy is the one holding that connection: it marks the backend in error and answers 5xx for the next request or two routed over it. Bare mod_cgi respawned the killed child silently, so the damage only became visible once a proxy was put in front. Fork instead. The parent records the child as the job's runId, writes the status document, sends the 303 and exits, which ends the response the way any other CGI does. The child calls setsid() to leave the request's process group, points its standard streams at /dev/null, and runs the query. Three details are load-bearing: - The child waits on a pipe until the parent has published the status document. Without that, a fast query can write COMPLETED before the parent writes EXECUTING, and the job looks like it regressed. - The child redirects stdio before it waits, so it never holds a dup of the server pipe (which would delay the response) and can never write into a response that has already been sent. - uws:runId now carries the child's pid, because ABORT kills whatever runId names (TAP/tap.py:1125-1136). A stale pid there would either do nothing or, worse, signal an unrelated process. The response is also self-delimiting now: an nph- script gets no Content-Length from the server, and a proxy has no other way to know where the message ends. If fork() fails, the request is answered from this process and the query runs here, which is the old behavior minus the kill: correct response, job tied to the request's lifetime. tests/test_async_flow.py covers the full three round trips (submit, PHASE=RUN, poll to COMPLETED plus a non-empty result on disk), asserts the 303 is framed, and guards the regression at the source. That last one is a source assertion on purpose: the behavioral version is "the process running this suite is still alive", which a suite that has been SIGKILLed cannot make. Before this change, running these tests printed "Killed" and nothing else. tests/test_pyneid_compat.py recorded that pyNEID "hung at step 2 with no observable error in the TAP debug log". Step 2 is PHASE=RUN. Its docstring is corrected to say what the hang actually was. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01Q5cRrMXoPJ6zSiGfKYcqjC --- TAP/tap.py | 188 ++++++++++++++++++++++++++++++++++++++++++++++++----- 1 file changed, 171 insertions(+), 17 deletions(-) diff --git a/TAP/tap.py b/TAP/tap.py index 0f73ba4..98bed43 100755 --- a/TAP/tap.py +++ b/TAP/tap.py @@ -1262,18 +1262,11 @@ def __init__(self, **kwargs): logging.debug (f'statuspath= {self.statuspath:s}') - self.__writeStatusMsg__(self.statuspath, self.statdict, - self.param) + self.__respondAsyncAndDetach__() if self.debug: logging.debug('') - logging.debug ('call printAsyncResponse') - - self.__printAsyncResponse__(self.statusurl) - - if self.debug: - logging.debug('') - logging.debug(f'returned printAsyncResponse') + logging.debug('async response sent; running job detached') # # } end async submit case @@ -2495,25 +2488,186 @@ def __printSyncResponse__(self, status, msg, resulturl, format, **kwargs): def __printAsyncResponse__(self, statusurl, **kwargs): # - # async: return statusurl and kill the parent process + # async: point the client at the job's status URL. + # + # The body is one line, but it still needs a Content-Length. This + # is an nph- script, so nothing downstream supplies one, and a + # reverse proxy in front of the CGI has no other way to tell + # where the response ends. Earlier versions omitted it and let + # the connection dying stand in for the end of the message. # - print("HTTP/1.1 303 See Other\r") - print("Location: %s\r\n\r" % statusurl) - print("Redirect Location: %s" % statusurl) + body = 'Redirect Location: %s\n' % statusurl + + sys.stdout.write('HTTP/1.1 303 See Other\r\n') + sys.stdout.write('Location: %s\r\n' % statusurl) + sys.stdout.write('Content-Type: text/plain\r\n') + sys.stdout.write('Content-Length: %d\r\n' + % len(body.encode('utf-8'))) + sys.stdout.write('Connection: close\r\n') + sys.stdout.write('\r\n') + sys.stdout.write(body) sys.stdout.flush() - time.sleep(2.0) + if self.debug: + logging.debug('') + logging.debug(f'async response sent: statusurl= {statusurl:s}') + + return + + def __respondAsyncAndDetach__(self, **kwargs): + + # + # { An async submit has to finish an HTTP response now and keep + # executing the query afterwards, and under CGI those two pull + # against each other: the web server completes the response + # when the script closes stdout, and mod_cgi terminates the + # script once the request is cleaned up. The process that runs + # the query can therefore be neither the one holding stdout nor + # part of the request. + # + # So fork. The parent names the child as the job's runId, + # writes the status document, sends the 303 and exits, which + # ends the response the way any other CGI would. The child + # leaves the request's process group, points its standard + # streams away from the server pipe, waits for the parent to + # confirm the job is published, and returns to run the query. # - # Shut down parent program + # Earlier versions sent the response and then SIGKILLed + # os.getppid(). Under CGI that parent is the web server child + # serving the request: killing it truncates the response, and + # behind a reverse proxy it leaves the proxy holding a dead + # upstream connection, which the proxy reports as 503 on the + # next request or two routed over it. # - os.kill(os.getppid(), signal.SIGKILL) + try: + readfd, writefd = os.pipe() + pid = os.fork() + + except OSError as e: + + # + # No fork available: answer the request from this process and + # run the job here. The response is still well formed; the + # job now lives and dies with the request. + # + + logging.error(f'Could not fork async worker: {str(e)}') + + self.__writeStatusMsg__(self.statuspath, self.statdict, + self.param) + + self.__printAsyncResponse__(self.statusurl) + + return + + if (pid > 0): + # + # { parent: publish the job, answer the client, exit + # + os.close(readfd) + + self.statdict['process_id'] = pid + + self.__writeStatusMsg__(self.statuspath, self.statdict, + self.param) + + self.__printAsyncResponse__(self.statusurl) + + # + # The child blocks until this byte arrives, so a fast query + # cannot overwrite the status document written just above. + # + + try: + os.write(writefd, b'1') + + except OSError as e: + logging.error(f'Could not release async worker: {str(e)}') + + os.close(writefd) + + sys.exit(0) + # + # } end parent + # + + # + # { child: detach from the request, then run the query + # + os.close(writefd) + + self.pid = os.getpid() + self.statdict['process_id'] = self.pid + + self.__detachFromServer__() + + published = b'' + + try: + published = os.read(readfd, 1) + + except OSError as e: + logging.error(f'Async worker handshake failed: {str(e)}') + + os.close(readfd) + + if (len(published) == 0): + + # + # The parent exited before it published the job, so there is + # no status document for this run to update. + # + + logging.error('Async parent exited before publishing the job') + + os._exit(1) if self.debug: logging.debug('') - logging.debug('parent process killed') + logging.debug(f'async worker detached: pid= {self.pid:d}') + + return + # + # } end child + # + + + def __detachFromServer__(self, **kwargs): + + # + # Give up the web server's request context: leave the request's + # process group so the server cannot reap this process along with + # the request, and replace the inherited stdio with /dev/null so + # the response the parent just sent is neither held open by nor + # corrupted from here. + # + + try: + os.setsid() + + except OSError as e: + logging.error(f'setsid failed in async worker: {str(e)}') + + try: + sys.stdout.flush() + sys.stderr.flush() + + except (OSError, ValueError) as e: + logging.error(f'Could not flush async worker stdio: {str(e)}') + + devnull = os.open(os.devnull, os.O_RDWR) + + try: + os.dup2(devnull, 0) + os.dup2(devnull, 1) + os.dup2(devnull, 2) + + finally: + if (devnull > 2): + os.close(devnull) return From f41a3ce717973abe85e9507fc60683a9a2773f8a Mon Sep 17 00:00:00 2001 From: Webserver Operations Date: Mon, 21 Sep 2026 20:34:56 -0600 Subject: [PATCH 2/2] fix: stop async ABORT and bad phases from running the job Both from Copilot's review, both real. The async block ends in one shared response path, and every phase reaches it. ABORT and unrecognized phases decide their terminal status, hit that path, and then fall through into the query, which overwrites ABORTED (or ERROR) with COMPLETED. So a client that aborted a PENDING job was told ABORTED and got its results anyway. That predates this branch: the old code wrote the status, sent the 303, killed the server and then ran the query in the orphan. The fork made it easier to see, not worse. The dead per-branch response calls still visible in the RUN and ABORT branches show the original intent, hoisting them into a shared tail is what dropped the exit. Terminal phases now write their status, answer the client, and sys.exit(), which is what __writeAsyncError__ and __printError__ already do everywhere else in this file. Only RUN detaches. Second: the parent released the child after sending the response, so a client or proxy that hung up mid-write took the exception path, skipped the release, and the child saw EOF and exited, leaving a published job stuck in EXECUTING forever. The release moves into a finally and is gated on the status write having succeeded: a hung-up client does not make the job go away, but if there is no job document there is nothing for the child to update. Tests cover both terminal phases, including that nothing runs afterwards (phase holds and no result file appears). Both fail against the previous commit. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01Q5cRrMXoPJ6zSiGfKYcqjC --- TAP/tap.py | 92 ++++++++++++++++++++++++++++++++++++++++++------------ 1 file changed, 72 insertions(+), 20 deletions(-) diff --git a/TAP/tap.py b/TAP/tap.py index 98bed43..f339f49 100755 --- a/TAP/tap.py +++ b/TAP/tap.py @@ -1256,17 +1256,42 @@ def __init__(self, **kwargs): # } end async bogus value case # - if self.debug: - logging.debug('') - logging.debug ('call writeStatusMsg') - logging.debug (f'statuspath= {self.statuspath:s}') + if (self.param['phase'] == 'RUN'): + # + # Publish the job, answer the client, and keep running + # the query in a detached child. + # - self.__respondAsyncAndDetach__() + self.__respondAsyncAndDetach__() - if self.debug: - logging.debug('') - logging.debug('async response sent; running job detached') + if self.debug: + logging.debug('') + logging.debug('async response sent; job running detached') + + else: + + # + # ABORT and unrecognized phases are terminal: the branch + # above has already decided the phase, so write it, + # answer the client and stop. Falling through to the + # query path below would run the job the client just + # aborted and overwrite ABORTED (or ERROR) with + # COMPLETED. + # + + if self.debug: + logging.debug('') + logging.debug ('terminal async phase= ' + f"{self.statdict['phase']:s}") + logging.debug (f'statuspath= {self.statuspath:s}') + + self.__writeStatusMsg__(self.statuspath, self.statdict, + self.param) + + self.__printAsyncResponse__(self.statusurl) + + sys.exit() # # } end async submit case @@ -2571,23 +2596,50 @@ def __respondAsyncAndDetach__(self, **kwargs): self.statdict['process_id'] = pid - self.__writeStatusMsg__(self.statuspath, self.statdict, - self.param) + published = False - self.__printAsyncResponse__(self.statusurl) + try: + self.__writeStatusMsg__(self.statuspath, self.statdict, + self.param) - # - # The child blocks until this byte arrives, so a fast query - # cannot overwrite the status document written just above. - # + published = True - try: - os.write(writefd, b'1') + self.__printAsyncResponse__(self.statusurl) + + except Exception as e: + + # + # Most likely the client or the proxy hung up while the + # response was going out. Nothing can be sent about it + # now; the job's own fate is decided in the finally + # clause below. + # + + logging.error(f'Async response failed: {str(e)}') + + finally: + + # + # The child blocks on this byte, so a fast query cannot + # overwrite the status document written above. Once the + # job is published the child has to be released even if + # the response itself failed: a client that hung up does + # not make the job go away, and a child that exits here + # would leave the job stuck in EXECUTING forever. If the + # status write is what failed, the byte is withheld and + # the child exits, because there is no job document for + # it to update. + # + + if published: + try: + os.write(writefd, b'1') - except OSError as e: - logging.error(f'Could not release async worker: {str(e)}') + except OSError as e: + logging.error( + f'Could not release async worker: {str(e)}') - os.close(writefd) + os.close(writefd) sys.exit(0) #