From 76015a748f07b38e1da5c6bf5809ba804c220642 Mon Sep 17 00:00:00 2001 From: Taus Date: Thu, 1 Oct 2026 12:14:42 +0000 Subject: [PATCH 1/2] Python: Normalise attribute ordering in `tsg-python` errors When decoding the `tsg-python` output fails, we currently output an error message such as ``` No context for node 27 in file /templates/test_module/tests/unit/modules/test_{{module_name}}.py with attributes {'_location': [8, 7, 8, 11], '_kind': 'Name', 'variable': 'salt'} ``` Here, the attribute dictionary contains valuable information for finding and diagnosing the issue (in the above example, for instance, the file contains weird template directives that make it not actually Python). Unfortunately, the order in which the attributes are output is not stable, which means a simple textual comparison of error messages is not enough to establish whether two errors are the same or not. To fix this, we now explicitly sort the attributes by key before outputting them. This makes the error output more stable, which should make it easier to see when it actually changes (as opposed to when it's the same error with a different attribute ordering. --- .../extractor/semmle/python/parser/tsg_parser.py | 15 +++++++++++++-- 1 file changed, 13 insertions(+), 2 deletions(-) diff --git a/python/extractor/semmle/python/parser/tsg_parser.py b/python/extractor/semmle/python/parser/tsg_parser.py index 1f77b648b2fa..66709ff6f205 100644 --- a/python/extractor/semmle/python/parser/tsg_parser.py +++ b/python/extractor/semmle/python/parser/tsg_parser.py @@ -147,6 +147,9 @@ def _decode_tsg_node_attributes(encoded_attrs, path, logger): ) return attrs +def _format_node_attributes(attrs): + return repr(dict(sorted(attrs.items()))) + def read_tsg_python_output(path, logger): command_args = tsg_command + [path] p = subprocess.Popen(command_args, stdout=subprocess.PIPE) @@ -203,7 +206,11 @@ def get_context(id, node_attr, path, logger): while "ctx" not in node_attr[id]: if "_inherited_ctx" not in node_attr[id]: - logger.error("No context for node {} in file {} with attributes {}\n".format(id, path, node_attr[id])) + logger.error( + "No context for node {} in file {} with attributes {}\n".format( + id, path, _format_node_attributes(node_attr[id]) + ) + ) # A missing context is most likely to be a "load", so return that. return ast.Load() id = node_attr[id]["_inherited_ctx"].id @@ -315,7 +322,11 @@ def parse(path, logger): nodes[id] = attrs["_is_literal"] continue if "_kind" not in attrs: - logger.error("Error: Graph node {} with attributes {} has no `_kind`!\n".format(id, attrs)) + logger.error( + "Error: Graph node {} with attributes {} has no `_kind`!\n".format( + id, _format_node_attributes(attrs) + ) + ) continue # This is not the node we are looking for (so don't bother creating it). if "_skip_to" in attrs: From 3ebbfc4693306fab19bb986cf9f02970bc5bd390 Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Thu, 1 Oct 2026 15:38:51 +0000 Subject: [PATCH 2/2] Python: Sort attributes in unknown-field diagnostics Co-authored-by: tausbn <1104778+tausbn@users.noreply.github.com> --- python/extractor/semmle/python/parser/tsg_parser.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/python/extractor/semmle/python/parser/tsg_parser.py b/python/extractor/semmle/python/parser/tsg_parser.py index 66709ff6f205..fe9d2224cada 100644 --- a/python/extractor/semmle/python/parser/tsg_parser.py +++ b/python/extractor/semmle/python/parser/tsg_parser.py @@ -364,7 +364,7 @@ def parse(path, logger): if field.startswith("_"): continue if field == "ctx": continue if field != "parenthesised" and field not in expected_fields: - logger.warning("Unknown field {} found among {} in node {}\n".format(field, attrs, id)) + logger.warning("Unknown field {} found among {} in node {}\n".format(field, _format_node_attributes(attrs), id)) # For fields that point to other AST nodes. if isinstance(val, Node):