diff --git a/changes-entries/sed-eos-on-error.txt b/changes-entries/sed-eos-on-error.txt new file mode 100644 index 00000000000..9a95f93dc3f --- /dev/null +++ b/changes-entries/sed-eos-on-error.txt @@ -0,0 +1,3 @@ + *) mod_sed: End the response properly when a sed script fails part-way + through it, rather than leaving the client waiting for a response + which never finishes. [Joe Orton] diff --git a/changes-entries/sed-l-command.txt b/changes-entries/sed-l-command.txt new file mode 100644 index 00000000000..3284b3fc23a --- /dev/null +++ b/changes-entries/sed-l-command.txt @@ -0,0 +1,3 @@ + *) mod_sed: Fix the octal escapes the "l" command writes for bytes with + the high bit set, which came out as nonsense such as "\/77" for 0xff. + [Joe Orton] diff --git a/docs/manual/mod/mod_sed.xml b/docs/manual/mod/mod_sed.xml index bd9308dd59e..a252fd8362a 100644 --- a/docs/manual/mod/mod_sed.xml +++ b/docs/manual/mod/mod_sed.xml @@ -108,8 +108,7 @@ page. OutputSed Sed command for filtering response content OutputSed sed-command -directory.htaccess - +directory

The OutputSed directive specifies the sed @@ -122,8 +121,7 @@ page. InputSed Sed command to filter request data (typically POST data) InputSed sed-command -directory.htaccess - +directory

The InputSed directive specifies the sed command diff --git a/modules/filters/mod_sed.c b/modules/filters/mod_sed.c index 193a41254ff..424dd0e1c07 100644 --- a/modules/filters/mod_sed.c +++ b/modules/filters/mod_sed.c @@ -54,6 +54,14 @@ typedef struct sed_filter_ctxt apr_size_t bufsize; apr_pool_t *tpool; int numbuckets; + /* Whether anything has been handed to the next filter yet. Once it + * has, a failure can no longer be turned into an error response. + */ + int passed; + /* Sticky failure of the evaluation, once the response is beyond + * rescue: the rest of the body is dropped but its metadata is not. + */ + apr_status_t evalerr; } sed_filter_ctxt; module AP_MODULE_DECLARE_DATA sed_module; @@ -119,6 +127,7 @@ static apr_status_t append_bucket(sed_filter_ctxt* ctx, char* buf, apr_size_t sz if (ctx->numbuckets >= MAX_TRANSIENT_BUCKETS) { b = apr_bucket_flush_create(ctx->r->connection->bucket_alloc); APR_BRIGADE_INSERT_TAIL(ctx->bb, b); + ctx->passed = 1; status = ap_pass_brigade(ctx->f->next, ctx->bb); apr_brigade_cleanup(ctx->bb); clear_ctxpool(ctx); @@ -272,11 +281,32 @@ static apr_status_t init_context(ap_filter_t *f, sed_expr_config *sed_cfg, int u return APR_SUCCESS; } +/* keep_metadata + * The response has been broken part-way through. Whatever is left in bb is + * unfiltered content which must not be sent, but its metadata -- above all + * the EOS -- still has to reach the next filter, or the response is never + * terminated. + */ +static void keep_metadata(sed_filter_ctxt *ctx, apr_bucket_brigade *bb) +{ + while (!APR_BRIGADE_EMPTY(bb)) { + apr_bucket *b = APR_BRIGADE_FIRST(bb); + APR_BUCKET_REMOVE(b); + if (APR_BUCKET_IS_METADATA(b)) { + APR_BRIGADE_INSERT_TAIL(ctx->bb, b); + } + else { + apr_bucket_destroy(b); + } + } +} + /* Entry function for Sed output filter */ static apr_status_t sed_response_filter(ap_filter_t *f, apr_bucket_brigade *bb) { apr_bucket *b; + apr_status_t rv; apr_status_t status = APR_SUCCESS; sed_config *cfg = ap_get_module_config(f->r->per_dir_config, &sed_module); @@ -291,8 +321,12 @@ static apr_status_t sed_response_filter(ap_filter_t *f, if (ctx == NULL) { + if (APR_BRIGADE_EMPTY(bb)) { + return ap_pass_brigade(f->next, bb); + } + if (APR_BUCKET_IS_EOS(APR_BRIGADE_FIRST(bb))) { - /* no need to run sed filter for Head requests */ + /* No body to filter */ ap_remove_output_filter(f); return ap_pass_brigade(f->next, bb); } @@ -305,6 +339,19 @@ static apr_status_t sed_response_filter(ap_filter_t *f, ctx->bb = apr_brigade_create(f->r->pool, f->c->bucket_alloc); } + else if (ctx->evalerr != APR_SUCCESS) { + /* The evaluation failed for an earlier brigade of this response, + * after some of it had already gone out. Nothing more can be + * filtered, but the metadata still has to be carried through so + * that the EOS ends the response. + */ + keep_metadata(ctx, bb); + if (!APR_BRIGADE_EMPTY(ctx->bb)) { + ap_pass_brigade(f->next, ctx->bb); + apr_brigade_cleanup(ctx->bb); + } + return ctx->evalerr; + } /* Here is the main logic. Iterate through all the buckets, read the * content of the bucket, call sed_eval_buffer on the data. @@ -328,49 +375,72 @@ static apr_status_t sed_response_filter(ap_filter_t *f, */ while (!APR_BRIGADE_EMPTY(bb)) { b = APR_BRIGADE_FIRST(bb); - if (APR_BUCKET_IS_EOS(b)) { - /* Now clean up the internal sed buffer */ - sed_finalize_eval(&ctx->eval, ctx); - status = flush_output_buffer(ctx); - if (status != APR_SUCCESS) { - break; + if (APR_BUCKET_IS_METADATA(b)) { + if (APR_BUCKET_IS_EOS(b)) { + /* Now clean up the internal sed buffer */ + sed_finalize_eval(&ctx->eval, ctx); } - /* Move the eos bucket to ctx->bb brigade */ - APR_BUCKET_REMOVE(b); - APR_BRIGADE_INSERT_TAIL(ctx->bb, b); - } - else if (APR_BUCKET_IS_FLUSH(b)) { + /* Flush what has been generated so far, so that the metadata + * keeps its place in the stream, and move it across. Buckets + * this filter has no opinion on are carried through rather + * than dropped: an error bucket has to reach the filters which + * act on it. + */ status = flush_output_buffer(ctx); if (status != APR_SUCCESS) { break; } - /* Move the flush bucket to ctx->bb brigade */ APR_BUCKET_REMOVE(b); APR_BRIGADE_INSERT_TAIL(ctx->bb, b); } else { - if (!APR_BUCKET_IS_METADATA(b)) { - const char *buf = NULL; - apr_size_t bytes = 0; + const char *buf = NULL; + apr_size_t bytes = 0; - status = apr_bucket_read(b, &buf, &bytes, APR_BLOCK_READ); - if (status == APR_SUCCESS) { - status = sed_eval_buffer(&ctx->eval, buf, bytes, ctx); - } - if (status != APR_SUCCESS) { - ap_log_rerror(APLOG_MARK, APLOG_ERR, status, f->r, APLOGNO(10394) "error evaluating sed on output"); - break; - } + status = apr_bucket_read(b, &buf, &bytes, APR_BLOCK_READ); + if (status == APR_SUCCESS) { + status = sed_eval_buffer(&ctx->eval, buf, bytes, ctx); + } + if (status != APR_SUCCESS) { + ap_log_rerror(APLOG_MARK, APLOG_ERR, status, f->r, APLOGNO(10394) "error evaluating sed on output"); + break; } apr_bucket_delete(b); } } + + /* Flush whatever did evaluate, even after a failure: it is valid output + * and the most of the response the client can still be given. + */ + rv = flush_output_buffer(ctx); if (status == APR_SUCCESS) { - status = flush_output_buffer(ctx); + status = rv; } + + if (status != APR_SUCCESS) { + if (!ctx->passed) { + /* None of the response has been written, so the caller can + * still turn this into a proper error response. Say nothing + * more than that it failed. + */ + apr_brigade_cleanup(ctx->bb); + clear_ctxpool(ctx); + return status; + } + /* Too late for that: part of the response is already on its way, + * so it has to be ended as the truncated response it has become. + * Dropping the EOS here instead leaves the header filters unrun + * and the client waiting for a response that never finishes. + */ + ctx->evalerr = status; + keep_metadata(ctx, bb); + } + if (!APR_BRIGADE_EMPTY(ctx->bb)) { + ctx->passed = 1; + rv = ap_pass_brigade(f->next, ctx->bb); if (status == APR_SUCCESS) { - status = ap_pass_brigade(f->next, ctx->bb); + status = rv; } apr_brigade_cleanup(ctx->bb); } diff --git a/modules/filters/regexp.c b/modules/filters/regexp.c index 4acccca6765..f07fee06eb4 100644 --- a/modules/filters/regexp.c +++ b/modules/filters/regexp.c @@ -535,6 +535,13 @@ static int _advance(char *lp, char *ep, step_vars_storage *vars) bbeg = vars->braslist[epint]; ct = vars->braelist[epint] - bbeg; ep++; + if (ct == 0) { + /* The capture matched nothing, so repeating it consumes + * nothing: it can only match once, here. Both loops below + * step lp by ct and would never make progress. + */ + continue; + } curlp = lp; while (ecmp(bbeg, lp, ct)) lp += ct; diff --git a/modules/filters/sed1.c b/modules/filters/sed1.c index 21d1cc78a50..f693a2a459d 100644 --- a/modules/filters/sed1.c +++ b/modules/filters/sed1.c @@ -73,6 +73,8 @@ static apr_status_t command(sed_eval_t *eval, sed_reptr_t *ipc, step_vars_storage *step_vars); static apr_status_t wline(sed_eval_t *eval, char *buf, apr_size_t sz); static apr_status_t arout(sed_eval_t *eval); +static void eval_errf(sed_eval_t *eval, const char *fmt, ...) + __attribute__((format(printf,2,3))); static void eval_errf(sed_eval_t *eval, const char *fmt, ...) { @@ -414,7 +416,7 @@ apr_status_t sed_eval_buffer(sed_eval_t *eval, const char *buf, apr_size_t bufsz /* Commands were not finalized properly. */ const char* error = sed_get_finalize_error(eval->commands, eval->pool); if (error) { - eval_errf(eval, error); + eval_errf(eval, "%s", error); return APR_EGENERAL; } } @@ -777,7 +779,12 @@ static apr_status_t command(sed_eval_t *eval, sed_reptr_t *ipc, switch(ipc->command) { case ACOM: - if (eval->aptr >= &eval->abuf[SED_ABUFSIZE]) { + /* One slot has to be left for the NULL which terminates abuf, + * or writing it runs off the end of the array and over aptr + * itself -- after which the next append writes through a NULL + * pointer. + */ + if (eval->aptr >= &eval->abuf[SED_ABUFSIZE - 1]) { eval_errf(eval, SEDERR_TMAMES, eval->lnum); } else { *eval->aptr++ = ipc; @@ -874,6 +881,13 @@ static apr_status_t command(sed_eval_t *eval, sed_reptr_t *ipc, continue; } if (!isprint(*p1 & 0377)) { + /* The three octal digits have to come off the byte + * value, not off a sign-extended char: 0xff is + * \377, and shifting it as a negative number gave + * "\/77". + */ + unsigned char uc = (unsigned char)*p1; + *p2++ = '\\'; if (p2 >= eval->lcomend) { *p2 = '\\'; @@ -883,7 +897,7 @@ static apr_status_t command(sed_eval_t *eval, sed_reptr_t *ipc, return rv; p2 = eval->genbuf; } - *p2++ = (*p1 >> 6) + '0'; + *p2++ = (uc >> 6) + '0'; if (p2 >= eval->lcomend) { *p2 = '\\'; rv = wline(eval, eval->genbuf, @@ -892,7 +906,7 @@ static apr_status_t command(sed_eval_t *eval, sed_reptr_t *ipc, return rv; p2 = eval->genbuf; } - *p2++ = ((*p1 >> 3) & 07) + '0'; + *p2++ = ((uc >> 3) & 07) + '0'; if (p2 >= eval->lcomend) { *p2 = '\\'; rv = wline(eval, eval->genbuf, @@ -901,7 +915,8 @@ static apr_status_t command(sed_eval_t *eval, sed_reptr_t *ipc, return rv; p2 = eval->genbuf; } - *p2++ = (*p1++ & 07) + '0'; + *p2++ = (uc & 07) + '0'; + p1++; if (p2 >= eval->lcomend) { *p2 = '\\'; rv = wline(eval, eval->genbuf, @@ -993,7 +1008,8 @@ static apr_status_t command(sed_eval_t *eval, sed_reptr_t *ipc, break; case RCOM: - if (eval->aptr >= &eval->abuf[SED_ABUFSIZE]) { + /* See ACOM: the terminating NULL needs a slot of its own. */ + if (eval->aptr >= &eval->abuf[SED_ABUFSIZE - 1]) { eval_errf(eval, SEDERR_TMRMES, eval->lnum); } else { *eval->aptr++ = ipc; diff --git a/test/modules/filters/env.py b/test/modules/filters/env.py index c78a8fac0b2..c0aba5f25c3 100644 --- a/test/modules/filters/env.py +++ b/test/modules/filters/env.py @@ -13,6 +13,7 @@ def __init__(self, env: 'HttpdTestEnv'): super().__init__(env=env) self.add_source_dir(os.path.dirname(inspect.getfile(FiltersTestSetup))) self.add_modules(["substitute", "sed"]) + self.add_cgi_module() class FiltersTestEnv(HttpdTestEnv): diff --git a/test/modules/filters/htdocs/test1/cgi/echo.py b/test/modules/filters/htdocs/test1/cgi/echo.py new file mode 100644 index 00000000000..985c38ec2c6 --- /dev/null +++ b/test/modules/filters/htdocs/test1/cgi/echo.py @@ -0,0 +1,12 @@ +#!/usr/bin/env python3 +# Echo the request body back verbatim, so an InputSed test can see exactly +# what the input filter handed to the handler. Read to EOF rather than +# CONTENT_LENGTH bytes: mod_sed rewrites the body but not the header, so the +# two need not agree. +import sys + +body = sys.stdin.buffer.read() +print("Content-Type: text/plain") +print() +sys.stdout.flush() +sys.stdout.buffer.write(body) diff --git a/test/modules/filters/htdocs/test1/cgi/nobody.py b/test/modules/filters/htdocs/test1/cgi/nobody.py new file mode 100644 index 00000000000..d2146834926 --- /dev/null +++ b/test/modules/filters/htdocs/test1/cgi/nobody.py @@ -0,0 +1,5 @@ +#!/usr/bin/env python3 +# Headers and nothing else, so the output filters see a response whose body +# is a lone EOS. +print("Content-Type: text/html") +print() diff --git a/test/modules/filters/test_003_sed.py b/test/modules/filters/test_003_sed.py new file mode 100644 index 00000000000..be402359c04 --- /dev/null +++ b/test/modules/filters/test_003_sed.py @@ -0,0 +1,358 @@ +import os +import re + +import pytest + +from pyhttpd.conf import HttpdConf + +# mod_sed implements the Solaris 10 sed language over the response (OutputSed) +# or the request body (InputSed). Almost none of it was covered before, so +# these tests walk the commands the module documents plus the paths mod_sed.c +# itself has: the 8000-byte output buffer, the transient-bucket flush, the +# empty-body short circuit and the error path. + +# The document most tests run against. Three lines, trailing newline. +DOC = "one monday two\nthree sunday four\nmonday monday monday\n" + +# huge.html: enough ordinary lines to push mod_sed past MAX_TRANSIENT_BUCKETS +# (50 buckets of MODSED_OUTBUF_SIZE, so 400000 bytes) and make it flush what it +# has to the client, then a single line past the 8 MB (MAX_BUF_SIZE) a line may +# grow to, which fails the evaluation. +HUGE_HEAD_LINES = 500 +HUGE_LINE_LEN = 900 +HUGE_HEAD_LEN = HUGE_HEAD_LINES * (HUGE_LINE_LEN + 1) +HUGE_TAIL_LEN = 9 * 1024 * 1024 + +# For a test whose failure mode is "never answers" rather than "answers +# wrongly". pyhttpd only gives curl --connect-timeout. +BOUNDED = ["--max-time", "30"] + + +class TestSed: + + @pytest.fixture(autouse=True, scope='class') + def _class_scope(self, env): + self.docdir = os.path.join(env.server_dir, "htdocs", "test1") + self.write(env, "sed.html", DOC) + # Same text with no newline after the last line. + self.write(env, "nonl.html", DOC[:-1]) + self.write(env, "empty.html", "") + # One line per 4 KB, well past MODSED_OUTBUF_SIZE and past the 50 + # transient buckets mod_sed flushes at. + self.write(env, "big.html", + "".join(f"line {i:06d} monday {'x' * 4000}\n" + for i in range(500))) + # A byte from each interesting class for the l command: printable, + # tab, DEL and three high-bit bytes. + self.write_bytes(env, "bytes.html", b"a\tb\x7fc\xffd\x80e\xa9f\n") + self.write(env, "insert.txt", "INSERTED\n") + # First line has nothing for \(a*\) to capture, second has "aa". + self.write(env, "backref.html", "b\naab\n") + head = "".join( + f"line {i:06d} monday ".ljust(HUGE_LINE_LEN, "y") + "\n" + for i in range(HUGE_HEAD_LINES)) + assert len(head) == HUGE_HEAD_LEN + self.write(env, "huge.html", head + "z" * HUGE_TAIL_LEN + "\n") + + @staticmethod + def write(env, name, text): + TestSed.write_bytes(env, name, text.encode()) + + @staticmethod + def write_bytes(env, name, data): + path = os.path.join(env.server_dir, "htdocs", "test1", name) + with open(path, "wb") as f: + f.write(data) + + @staticmethod + def sed_path(env, name): + """The path of a document as an "r" command has to spell it: sed + reads a backslash in a filename as an escape and drops it, so a + Windows path only survives with forward slashes.""" + return os.path.join(env.server_dir, "htdocs", "test1", + name).replace(os.sep, "/") + + def configure(self, env, exprs, extra="", input_sed=False): + if isinstance(exprs, str): + exprs = [exprs] + directive = "InputSed" if input_sed else "OutputSed" + lines = "\n".join(f' {directive} "{e}"' for e in exprs) + conf = HttpdConf(env, extras={ + f"test1.{env.http_tld}": f""" + + AddOutputFilterByType SED text/html + AddInputFilter SED py + AddHandler cgi-script .py + Options +ExecCGI +{lines} + + {extra} + """, + }) + conf.add_vhost_test1() + conf.install() + assert env.apache_restart() == 0 + + def get(self, env, path="/sed.html", options=None): + return env.curl_get(env.mkurl("http", "test1", path), options=options) + + def body(self, env, exprs, path="/sed.html", **kwargs): + self.configure(env, exprs, **kwargs) + r = self.get(env, path) + assert r.response, f"no response for {exprs}" + assert r.response["status"] == 200, \ + f"{exprs}: status {r.response['status']}" + return r.response["body"] + + # --- substitution ----------------------------------------------------- + + # s/// replaces the first match on each line and nothing else. + def test_filters_003_01(self, env): + assert self.body(env, "s/monday/MON/").decode() == \ + "one MON two\nthree sunday four\nMON monday monday\n" + + # s///g replaces every match on the line. + def test_filters_003_02(self, env): + assert self.body(env, "s/monday/MON/g").decode() == \ + "one MON two\nthree sunday four\nMON MON MON\n" + + # s///2 replaces only the second match. + def test_filters_003_03(self, env): + assert self.body(env, "s/monday/MON/2").decode() == \ + "one monday two\nthree sunday four\nmonday MON monday\n" + + # Several OutputSed directives accumulate into one script, applied in the + # order they appear. This is the configuration the manual shows. + def test_filters_003_04(self, env): + assert self.body(env, ["s/monday/MON/g", "s/sunday/SUN/g"]).decode() \ + == "one MON two\nthree SUN four\nMON MON MON\n" + + # A back reference in the replacement. + def test_filters_003_05(self, env): + assert self.body(env, r"s/\(mon\)day/[\1]/g").decode() == \ + "one [mon] two\nthree sunday four\n[mon] [mon] [mon]\n" + + # & in the replacement is the whole match. + def test_filters_003_06(self, env): + assert self.body(env, "s/monday/<&>/").decode() == \ + "one two\nthree sunday four\n monday monday\n" + + # --- addresses -------------------------------------------------------- + + # A line-number address. + def test_filters_003_07(self, env): + assert self.body(env, "2d").decode() == \ + "one monday two\nmonday monday monday\n" + + # $ is the last line. + def test_filters_003_08(self, env): + assert self.body(env, "$d").decode() == \ + "one monday two\nthree sunday four\n" + + # A regex range. + def test_filters_003_09(self, env): + assert self.body(env, "/sunday/,$d").decode() == "one monday two\n" + + # A negated address. + def test_filters_003_10(self, env): + assert self.body(env, "/sunday/!d").decode() == "three sunday four\n" + + # = writes the line number ahead of the line. + def test_filters_003_11(self, env): + assert self.body(env, "=").decode() == \ + "1\none monday two\n2\nthree sunday four\n3\nmonday monday monday\n" + + # --- hold space ------------------------------------------------------- + + # G appends the hold space, which starts empty: classic double spacing. + def test_filters_003_12(self, env): + assert self.body(env, "G").decode() == \ + "one monday two\n\nthree sunday four\n\nmonday monday monday\n\n" + + # h saves, then x swaps: every line is replaced by the one before it. + def test_filters_003_13(self, env): + assert self.body(env, "x").decode() == \ + "\none monday two\nthree sunday four\n" + + # H;$!d;x collects the whole document in the hold space and prints it once. + def test_filters_003_14(self, env): + assert self.body(env, ["H", "$!d", "x"]).decode() == "\n" + DOC + + # --- transliteration -------------------------------------------------- + + def test_filters_003_15(self, env): + assert self.body(env, "y/aeiou/AEIOU/").decode() == \ + "OnE mOndAy twO\nthrEE sUndAy fOUr\nmOndAy mOndAy mOndAy\n" + + # --- line handling ---------------------------------------------------- + + # A document whose last line has no newline gets one: documented Solaris + # sed behaviour, and the reason mod_sed drops Content-Length. + def test_filters_003_16(self, env): + assert self.body(env, "s/monday/MON/g", path="/nonl.html").decode() \ + == "one MON two\nthree sunday four\nMON MON MON\n" + + # An empty body stays empty; mod_sed must not invent a line. + def test_filters_003_17(self, env): + assert self.body(env, "s/monday/MON/", path="/empty.html") == b"" + + # mod_sed drops the handler's Content-Length because the body it produces + # is a different size. The length that reaches the client must be the one + # after filtering -- ap_content_length_filter recomputes it -- and never + # the file's own. + def test_filters_003_18(self, env): + self.configure(env, "s/monday/MONDAYMONDAY/g") + r = self.get(env) + assert r.response["status"] == 200 + grown = len(DOC) + 4 * len("MONDAY") + assert len(r.response["body"]) == grown + assert r.response["header"]["content-length"] == str(grown), \ + r.response["header"] + + # The handler still generates a body for HEAD -- the protocol filters + # discard it -- so the filter runs and HEAD must announce the same + # Content-Length the matching GET returns, not the file's own. + def test_filters_003_19(self, env): + self.configure(env, "s/monday/MONDAYMONDAY/g") + r = self.get(env, options=["-I"]) + assert r.response, "no response to HEAD" + assert r.response["status"] == 200 + grown = len(DOC) + 4 * len("MONDAY") + assert r.response["header"]["content-length"] == str(grown), \ + r.response["header"] + + # A handler which produces headers and no body reaches mod_sed as a lone + # EOS. There is nothing to evaluate, so the filter takes itself out of + # the chain without touching the response. + def test_filters_003_20(self, env): + self.configure(env, "s/monday/MON/") + r = self.get(env, "/cgi/nobody.py") + assert r.response, "no response" + assert r.response["status"] == 200 + assert r.response["body"] == b"" + + # A body far larger than the 8000-byte output buffer, and with more than + # the 50 transient buckets mod_sed flushes at, comes back intact. + def test_filters_003_21(self, env): + body = self.body(env, "s/monday/MON/", path="/big.html") + assert len(body) == 500 * (len("line 000000 MON ") + 4000 + 1) + lines = body.decode().splitlines() + assert len(lines) == 500 + assert lines[0] == "line 000000 MON " + "x" * 4000 + assert lines[499] == "line 000499 MON " + "x" * 4000 + + # --- the l command ---------------------------------------------------- + + # l writes the line in a printable form: control characters as the escapes + # in trans[], everything else non-printable as a three-digit octal escape. + # Bytes with the top bit set used to come out as nonsense ("\/77" for 0xff) + # because the shift was done on a signed char. + def test_filters_003_22(self, env): + body = self.body(env, "l;d", path="/bytes.html") + assert body == b"a\\11b\\177c\\377d\\200e\\251f\n", body + + # --- error handling --------------------------------------------------- + + # A branch to a label that is never defined compiles -- the label could + # still arrive from a later OutputSed -- and fails at the first byte of + # every response instead. Nothing has been written at that point, so the + # failure can and must still become a 500. + def test_filters_003_23(self, env): + env.httpd_error_log.add_ignored_lognos(["AH02998", "AH10394"]) + self.configure(env, "b nolabel") + r = self.get(env) + assert r.response, "no response at all" + assert r.response["status"] == 500, r.response["status"] + assert env.httpd_error_log.scan_recent( + re.compile(r'.*undefined label: nolabel')) + + # The same failure once the response is already on its way. huge.html is + # 450 KB of ordinary lines -- enough for mod_sed to hit MAX_TRANSIENT_BUCKETS + # and flush the start of the response to the client -- followed by one line + # over the 8 MB a line may grow to, which fails the evaluation. + # + # The response cannot become a 500 any more, so it has to be terminated as + # what it is: a truncated 200. mod_sed used to skip both the final flush + # and the pass because of the error status, dropping the EOS with them, and + # the client was left waiting on a response that never ended. + def test_filters_003_24(self, env): + env.httpd_error_log.add_ignored_lognos(["AH02998", "AH10394"]) + self.configure(env, "s/monday/MON/") + r = self.get(env, "/huge.html") + assert r.exit_code == 0, \ + f"curl failed ({r.exit_code}): the response was never terminated" + assert r.response["status"] == 200 + body = r.response["body"] + # Truncated: the lines evaluated before the failure, and no more. + assert 0 < len(body) < HUGE_HEAD_LEN + 1024, len(body) + assert body.startswith(b"line 000000 MON ") + assert env.httpd_error_log.scan_recent( + re.compile(r'.*error evaluating sed on output')) + + # SED_ABUFSIZE is the size of the append/read queue a line may build up. + # Filling it exactly used to write the terminating NULL one past the end + # of eval->abuf, over eval->aptr itself, and the next line then wrote + # through that NULL pointer. + def test_filters_003_25(self, env): + env.httpd_error_log.add_ignored_lognos(["AH02998"]) + path = self.sed_path(env, "insert.txt") + self.configure(env, [f"r {path}"] * 20) + r = self.get(env) + assert r.response, "no response: the server died" + assert r.response["status"] == 200 + # 19 of the 20 fit; the last is refused, and says so. + assert r.response["body"].count(b"INSERTED") == 3 * 19 + assert env.httpd_error_log.scan_recent( + re.compile(r'.*too many reads after line')) + + # One below the limit has always worked and must keep working. + def test_filters_003_26(self, env): + path = self.sed_path(env, "insert.txt") + body = self.body(env, [f"r {path}"] * 19) + assert body.count(b"INSERTED") == 3 * 19 + + # --- backreferences --------------------------------------------------- + + # A starred backreference whose capture matched nothing repeats a + # zero-length match: the matcher steps the input pointer by the capture + # length, so it has to stop rather than repeat it, and the star can only + # ever match once. Both lines reduce to their first match plus the rest. + # + # --max-time because a regression here does not return an error, it stops + # answering: pyhttpd gives curl --connect-timeout but no overall limit, so + # without one of our own a failure hangs the whole run rather than failing + # it. Note the worker keeps spinning after curl gives up. + def test_filters_003_27(self, env): + self.configure(env, r"s/\(a*\)\1*/X/") + r = self.get(env, "/backref.html", options=BOUNDED) + assert r.exit_code == 0, f"curl failed ({r.exit_code}): no response" + assert r.response["status"] == 200 + assert r.response["body"] == b"Xb\nXb\n" + + # --- input filter ----------------------------------------------------- + + # The same expression on a request body, which runs the same matcher. + def test_filters_003_28(self, env): + self.configure(env, r"s/\(a*\)\1*/X/", input_sed=True) + r = env.curl_post_data(env.mkurl("http", "test1", "/cgi/echo.py"), + data="b\naab\n", options=list(BOUNDED)) + assert r.exit_code == 0, f"curl failed ({r.exit_code}): no response" + assert r.response["status"] == 200 + assert r.response["body"] == b"Xb\nXb\n" + + # InputSed rewrites the request body before the handler sees it. + def test_filters_003_29(self, env): + self.configure(env, "s/monday/MON/g", input_sed=True) + r = env.curl_post_data(env.mkurl("http", "test1", "/cgi/echo.py"), + data="one monday two\nmonday monday\n") + assert r.response, "no response" + assert r.response["status"] == 200 + assert r.response["body"] == b"one MON two\nMON MON\n" + + # A request body with no trailing newline gets one, as the manual warns. + def test_filters_003_30(self, env): + self.configure(env, "s/monday/MON/g", input_sed=True) + r = env.curl_post_data(env.mkurl("http", "test1", "/cgi/echo.py"), + data="one monday two") + assert r.response, "no response" + assert r.response["body"] == b"one MON two\n" diff --git a/test/pyhttpd/env.py b/test/pyhttpd/env.py index 02d82ecae25..88ba7adda18 100644 --- a/test/pyhttpd/env.py +++ b/test/pyhttpd/env.py @@ -84,6 +84,20 @@ def add_modules(self, modules: List[str]): def add_optional_modules(self, modules: List[str]): self._optional_modules.extend(modules) + def _cgi_module(self): + """The name of a CGI module this build has: mod_cgid for preference, + mod_cgi otherwise. There is no mod_cgid on Windows.""" + if sys.platform != "win32" \ + and os.path.isfile(os.path.join(self.env.libexec_dir, + "mod_cgid.so")): + return "cgid" + return "cgi" + + def add_cgi_module(self): + """Load a CGI module, for a suite which needs CGI but does not care + which of mod_cgid and mod_cgi runs it.""" + self.add_modules([self._cgi_module()]) + def make(self): self._make_dirs() self._make_conf() @@ -153,8 +167,8 @@ def _make_modules_conf(self): # issue load directives for all modules we want that are shared missing_mods = list() modules = self._modules - if sys.platform == "win32": - modules = ["cgi" if m == "cgid" else m for m in modules] + modules = [self._cgi_module() if m == "cgid" else m + for m in modules] for m in modules: match = re.match(r'^mod_(.+)$', m) if match: