From 42c63211efac2f3ef55528d2cd513adbbe289ba8 Mon Sep 17 00:00:00 2001 From: Thiago Lugarini Date: Tue, 18 Aug 2026 16:13:55 -0300 Subject: [PATCH] feat(amwscan): map the Exploit vocabulary from the engine's own words MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit #96 made the parenthesised token the discriminator. The Function tokens were already in the table; the Exploit ones were not, and on the account this came from that is 779 findings, 732 of them `execution` alone, all landing on other/medium/heuristic. AMWScan prints what each pattern means, on the "- " line under every finding. Each mapping quotes that description beside it, so the next person can check the mapping against the source rather than against my reading of a token name: execution "RCE ... execute PHP code on the target machine via HTTP" -> backdoor / critical, the same thing eval means nano "a family of PHP webshells ... code golfed to be stealthy" -> webshell / critical clever_include "LFI ... inject and execute arbitrary commands or code" -> injection / high, matching the existing `include` entry infected_comment "comments composed by 5 random chars usually used to detect if a file is infected yet" -> other / high: a marker something left behind, not a technique the file performs The obfuscation family — base64_long, hex_char, double_var2, concat_vars_array, concat_vars_with_spaces — sits at MEDIUM, matching the existing `encoded` entry rather than `obfuscated`. Every one of their descriptions says the technique is USUALLY used for malicious code, and usually is the operative word: minified libraries, licence blobs and legitimate encoders trip the same patterns. A heuristic that fires on ordinary vendor code at high severity is one whose severity stops meaning anything. Two are deliberately left unmapped, with the reasoning written beside the ones that are: etc_passwd true of an attacker, and also of every config parser, test fixture and tutorial that names the path php_uname one information-gathering call that installers make legitimately; the engine files it under RCE, which is more than a single call earns They stay unknown, which means other/medium and a line in the note counts, so they keep showing up as something to decide about. An unmapped rule is not a discarded one. Severity does not feed the score — confidence does, and all of these are heuristic — so no verdict moves. What moves is what a person reads when deciding which of 779 findings to look at first. Confirmed by mutation in both directions: removing `execution` fails with 'category "other", wanted "backdoor"', and raising hex_char to high fails with 'obfuscation alone is not a reason to act'. --- internal/adapter/amwscan/rules.go | 54 ++++++++++++ internal/adapter/amwscan/rules_test.go | 85 +++++++++++++++++++ .../amwscan_exploit_vocabulary_test.go | 59 +++++++++++++ 3 files changed, 198 insertions(+) create mode 100644 internal/adapter/amwscan/rules_test.go create mode 100644 tests/contract/amwscan_exploit_vocabulary_test.go diff --git a/internal/adapter/amwscan/rules.go b/internal/adapter/amwscan/rules.go index 401fe59..a6b881f 100644 --- a/internal/adapter/amwscan/rules.go +++ b/internal/adapter/amwscan/rules.go @@ -61,6 +61,60 @@ var ruleTable = map[string]mapping{ "suspicious": {schema.CategoryOther, schema.SeverityMedium, schema.ConfidenceHeuristic}, "unsafe": {schema.CategoryOther, schema.SeverityMedium, schema.ConfidenceHeuristic}, "permissions": {schema.CategorySuspiciousPerms, schema.SeverityMedium, schema.ConfidenceAnomaly}, + + // AMWScan's Exploit vocabulary, mapped from the descriptions the engine itself prints + // on the "- " line under every finding. Those descriptions are quoted here so the next + // person can check the mapping against the source rather than against my reading of it. + // Captured from a live account (see tests/testdata/raw/amwscan/PROVENANCE.md), where + // `execution` alone accounted for 732 findings. + // + // "RCE (Remote Code Execution) allow remote attackers to execute PHP code on the + // target machine via HTTP" — the same thing `eval` means, which is why it lands in the + // same place. + "execution": {schema.CategoryBackdoor, schema.SeverityCritical, schema.ConfidenceHeuristic}, + // "Nano is a family of PHP webshells which are code golfed to be extremely stealthy + // and efficient" — a named webshell family, not a pattern that happens to look bad. + "nano": {schema.CategoryWebshell, schema.SeverityCritical, schema.ConfidenceHeuristic}, + // "LFI (Local File Inclusion), through a image inclusion, allow remote attackers to + // inject and execute arbitrary commands or code" — the same category and weight the + // table already gives plain `include`. + "clever_include": {schema.CategoryInjection, schema.SeverityHigh, schema.ConfidenceHeuristic}, + + // The obfuscation family. Every one of these says "usually used for the obfuscation of + // malicious code", and USUALLY is the operative word: minified libraries, licence + // blobs and legitimate encoders trip them too. + // + // Medium, matching the existing `encoded` entry rather than `obfuscated`. Obfuscation + // on its own is a reason to look, not a reason to act, and a heuristic that fires on + // ordinary vendor code at high severity is one whose severity stops meaning anything. + "base64_long": {schema.CategoryObfuscation, schema.SeverityMedium, schema.ConfidenceHeuristic}, + "hex_char": {schema.CategoryObfuscation, schema.SeverityMedium, schema.ConfidenceHeuristic}, + "double_var2": {schema.CategoryObfuscation, schema.SeverityMedium, schema.ConfidenceHeuristic}, + "concat_vars_array": {schema.CategoryObfuscation, schema.SeverityMedium, schema.ConfidenceHeuristic}, + "concat_vars_with_spaces": {schema.CategoryObfuscation, schema.SeverityMedium, schema.ConfidenceHeuristic}, + + // "Comments composed by 5 random chars usually used to detect if a file is infected + // yet" — a marker malware leaves behind to recognise its own work. Not a technique the + // file performs, so it is not backdoor or webshell; it is a strong sign the file has + // been touched by something that keeps bookkeeping. + "infected_comment": {schema.CategoryOther, schema.SeverityHigh, schema.ConfidenceHeuristic}, + + // Two are deliberately NOT here, and the reason belongs next to the ones that are. + // + // etc_passwd — "the /etc/passwd file on Unix systems contains password information, + // an attacker who has accessed the etc/passwd file may attempt a brute force + // attack". True of an attacker; also true of every config parser, test fixture and + // tutorial that mentions the path. The engine is describing what the string means + // when an attacker wrote it, and the pattern cannot tell who did. + // + // php_uname — the engine files it under RCE. It is one information-gathering call, + // which diagnostics and installers make legitimately. Calling that critical on its + // own would put ordinary code next to webshells in the same list. + // + // Both stay unknown, which means other/medium/heuristic and a line in the note counts + // so they keep showing up as something to decide about. That is the honest state for a + // pattern whose meaning depends on who wrote the file, and an unmapped rule is not a + // discarded one. } // classify translates the engine's rule name into the normalized schema. diff --git a/internal/adapter/amwscan/rules_test.go b/internal/adapter/amwscan/rules_test.go new file mode 100644 index 0000000..4c31e48 --- /dev/null +++ b/internal/adapter/amwscan/rules_test.go @@ -0,0 +1,85 @@ +package amwscan + +import ( + "testing" + + "github.com/thiagoluga/SentinelHost/internal/schema" +) + +// Obfuscation is a reason to look, not a reason to act. +// +// Every entry in that family says the technique is USUALLY used for malicious code, and +// usually is the operative word — minified libraries, licence blobs and legitimate +// encoders trip the same patterns. They sit at medium, matching the existing `encoded` +// entry, because a heuristic that fires on ordinary vendor code at high severity is one +// whose severity stops meaning anything. +func TestTheObfuscationFamilyStaysAtMedium(t *testing.T) { + for _, token := range []string{ + "base64_long", "hex_char", "double_var2", + "concat_vars_array", "concat_vars_with_spaces", + } { + m, known := classify("Exploit", token) + if !known { + t.Errorf("%s is not in the table", token) + continue + } + if m.category != schema.CategoryObfuscation { + t.Errorf("%s: category %q, wanted obfuscation", token, m.category) + } + if m.severity != schema.SeverityMedium { + t.Errorf("%s: severity %q, wanted medium — obfuscation alone is not a reason "+ + "to act, and inflating it costs the severity field its meaning", token, m.severity) + } + } +} + +// The two deliberate omissions stay unknown, and unknown is not discarded. +// +// etc_passwd matches every config parser and tutorial that names the path; php_uname is +// one information-gathering call that installers make legitimately. The engine describes +// what they mean when an attacker wrote them, and the pattern cannot tell who did. They +// come out as other/medium and get counted, so they keep showing up as something to +// decide about rather than quietly becoming a verdict. +func TestTheAmbiguousPatternsAreLeftUnknownOnPurpose(t *testing.T) { + for _, token := range []string{"etc_passwd", "php_uname"} { + m, known := classify("Exploit", token) + if known { + t.Errorf("%s was mapped to %+v. If that is deliberate, the reasoning written "+ + "beside it in rules.go has to change too — it currently says the opposite", token, m) + } + if m.category != schema.CategoryOther || m.severity != schema.SeverityMedium { + t.Errorf("%s fell back to %q/%q, wanted other/medium", token, m.category, m.severity) + } + } +} + +// The token beats the rule name, and a hash falls through to it. +// +// The whole point of D-057: Function (eval) is a backdoor because of `eval`, not because +// of "Function"; Signature (11413268) is known malware because of "Signature", since no +// table will ever hold that hash. +func TestTheTokenIsTriedBeforeTheRuleName(t *testing.T) { + cases := []struct { + rule, token string + category schema.Category + confidence schema.Confidence + }{ + {"Function", "eval", schema.CategoryBackdoor, schema.ConfidenceHeuristic}, + {"Function", "exec", schema.CategoryWebshell, schema.ConfidenceHeuristic}, + {"Exploit", "execution", schema.CategoryBackdoor, schema.ConfidenceHeuristic}, + {"Signature", "11413268", schema.CategoryKnownMalware, schema.ConfidenceSignature}, + } + for _, tc := range cases { + m, known := classify(tc.rule, tc.token) + if !known { + t.Errorf("%s (%s): not classified", tc.rule, tc.token) + continue + } + if m.category != tc.category { + t.Errorf("%s (%s): category %q, wanted %q", tc.rule, tc.token, m.category, tc.category) + } + if m.confidence != tc.confidence { + t.Errorf("%s (%s): confidence %q, wanted %q", tc.rule, tc.token, m.confidence, tc.confidence) + } + } +} diff --git a/tests/contract/amwscan_exploit_vocabulary_test.go b/tests/contract/amwscan_exploit_vocabulary_test.go new file mode 100644 index 0000000..b7ba9c2 --- /dev/null +++ b/tests/contract/amwscan_exploit_vocabulary_test.go @@ -0,0 +1,59 @@ +package contract_test + +import ( + "strings" + "testing" + + "github.com/thiagoluga/SentinelHost/internal/adapter/amwscan" + "github.com/thiagoluga/SentinelHost/internal/schema" +) + +// The Exploit vocabulary is mapped from what the engine says each pattern means. +// +// AMWScan prints a description under every finding, and those descriptions are quoted in +// the rule table next to each entry so the mapping can be checked against the source +// rather than against somebody's reading of a token name: +// +// execution "RCE (Remote Code Execution) allow remote attackers to execute PHP code" +// hex_char "Hex char is usually used for the obfuscation of malicious code" +// +// Before this, every one of them landed on other/medium/heuristic — 732 findings of +// `execution` alone on the account this fixture came from. +func TestTheExploitVocabularyIsClassified(t *testing.T) { + a := amwscan.New().WithStat(fakeStat) + + rep, err := a.Parse(raw(amwscan.Slug, + fixture(t, "amwscan", "real-0.15.1-function-and-exploit.txt"), schema.StatusCompleted)) + if err != nil { + t.Fatalf("Parse: %v", err) + } + + want := map[string]struct { + category schema.Category + severity schema.Severity + }{ + "execution": {schema.CategoryBackdoor, schema.SeverityCritical}, + "hex_char": {schema.CategoryObfuscation, schema.SeverityMedium}, + } + + seen := map[string]bool{} + for _, f := range rep.Findings { + for token, exp := range want { + if !strings.Contains(f.MatchedContent, "("+token+")") { + continue + } + seen[token] = true + if f.Category != exp.category { + t.Errorf("%s: category %q, wanted %q", token, f.Category, exp.category) + } + if f.Severity != exp.severity { + t.Errorf("%s: severity %q, wanted %q", token, f.Severity, exp.severity) + } + } + } + for token := range want { + if !seen[token] { + t.Errorf("the fixture carries no %q finding; it is the evidence this mapping rests on", token) + } + } +}