diff --git a/internal/adapter/amwscan/rules.go b/internal/adapter/amwscan/rules.go index 401fe59..a6b881f 100644 --- a/internal/adapter/amwscan/rules.go +++ b/internal/adapter/amwscan/rules.go @@ -61,6 +61,60 @@ var ruleTable = map[string]mapping{ "suspicious": {schema.CategoryOther, schema.SeverityMedium, schema.ConfidenceHeuristic}, "unsafe": {schema.CategoryOther, schema.SeverityMedium, schema.ConfidenceHeuristic}, "permissions": {schema.CategorySuspiciousPerms, schema.SeverityMedium, schema.ConfidenceAnomaly}, + + // AMWScan's Exploit vocabulary, mapped from the descriptions the engine itself prints + // on the "- " line under every finding. Those descriptions are quoted here so the next + // person can check the mapping against the source rather than against my reading of it. + // Captured from a live account (see tests/testdata/raw/amwscan/PROVENANCE.md), where + // `execution` alone accounted for 732 findings. + // + // "RCE (Remote Code Execution) allow remote attackers to execute PHP code on the + // target machine via HTTP" — the same thing `eval` means, which is why it lands in the + // same place. + "execution": {schema.CategoryBackdoor, schema.SeverityCritical, schema.ConfidenceHeuristic}, + // "Nano is a family of PHP webshells which are code golfed to be extremely stealthy + // and efficient" — a named webshell family, not a pattern that happens to look bad. + "nano": {schema.CategoryWebshell, schema.SeverityCritical, schema.ConfidenceHeuristic}, + // "LFI (Local File Inclusion), through a image inclusion, allow remote attackers to + // inject and execute arbitrary commands or code" — the same category and weight the + // table already gives plain `include`. + "clever_include": {schema.CategoryInjection, schema.SeverityHigh, schema.ConfidenceHeuristic}, + + // The obfuscation family. Every one of these says "usually used for the obfuscation of + // malicious code", and USUALLY is the operative word: minified libraries, licence + // blobs and legitimate encoders trip them too. + // + // Medium, matching the existing `encoded` entry rather than `obfuscated`. Obfuscation + // on its own is a reason to look, not a reason to act, and a heuristic that fires on + // ordinary vendor code at high severity is one whose severity stops meaning anything. + "base64_long": {schema.CategoryObfuscation, schema.SeverityMedium, schema.ConfidenceHeuristic}, + "hex_char": {schema.CategoryObfuscation, schema.SeverityMedium, schema.ConfidenceHeuristic}, + "double_var2": {schema.CategoryObfuscation, schema.SeverityMedium, schema.ConfidenceHeuristic}, + "concat_vars_array": {schema.CategoryObfuscation, schema.SeverityMedium, schema.ConfidenceHeuristic}, + "concat_vars_with_spaces": {schema.CategoryObfuscation, schema.SeverityMedium, schema.ConfidenceHeuristic}, + + // "Comments composed by 5 random chars usually used to detect if a file is infected + // yet" — a marker malware leaves behind to recognise its own work. Not a technique the + // file performs, so it is not backdoor or webshell; it is a strong sign the file has + // been touched by something that keeps bookkeeping. + "infected_comment": {schema.CategoryOther, schema.SeverityHigh, schema.ConfidenceHeuristic}, + + // Two are deliberately NOT here, and the reason belongs next to the ones that are. + // + // etc_passwd — "the /etc/passwd file on Unix systems contains password information, + // an attacker who has accessed the etc/passwd file may attempt a brute force + // attack". True of an attacker; also true of every config parser, test fixture and + // tutorial that mentions the path. The engine is describing what the string means + // when an attacker wrote it, and the pattern cannot tell who did. + // + // php_uname — the engine files it under RCE. It is one information-gathering call, + // which diagnostics and installers make legitimately. Calling that critical on its + // own would put ordinary code next to webshells in the same list. + // + // Both stay unknown, which means other/medium/heuristic and a line in the note counts + // so they keep showing up as something to decide about. That is the honest state for a + // pattern whose meaning depends on who wrote the file, and an unmapped rule is not a + // discarded one. } // classify translates the engine's rule name into the normalized schema. diff --git a/internal/adapter/amwscan/rules_test.go b/internal/adapter/amwscan/rules_test.go new file mode 100644 index 0000000..4c31e48 --- /dev/null +++ b/internal/adapter/amwscan/rules_test.go @@ -0,0 +1,85 @@ +package amwscan + +import ( + "testing" + + "github.com/thiagoluga/SentinelHost/internal/schema" +) + +// Obfuscation is a reason to look, not a reason to act. +// +// Every entry in that family says the technique is USUALLY used for malicious code, and +// usually is the operative word — minified libraries, licence blobs and legitimate +// encoders trip the same patterns. They sit at medium, matching the existing `encoded` +// entry, because a heuristic that fires on ordinary vendor code at high severity is one +// whose severity stops meaning anything. +func TestTheObfuscationFamilyStaysAtMedium(t *testing.T) { + for _, token := range []string{ + "base64_long", "hex_char", "double_var2", + "concat_vars_array", "concat_vars_with_spaces", + } { + m, known := classify("Exploit", token) + if !known { + t.Errorf("%s is not in the table", token) + continue + } + if m.category != schema.CategoryObfuscation { + t.Errorf("%s: category %q, wanted obfuscation", token, m.category) + } + if m.severity != schema.SeverityMedium { + t.Errorf("%s: severity %q, wanted medium — obfuscation alone is not a reason "+ + "to act, and inflating it costs the severity field its meaning", token, m.severity) + } + } +} + +// The two deliberate omissions stay unknown, and unknown is not discarded. +// +// etc_passwd matches every config parser and tutorial that names the path; php_uname is +// one information-gathering call that installers make legitimately. The engine describes +// what they mean when an attacker wrote them, and the pattern cannot tell who did. They +// come out as other/medium and get counted, so they keep showing up as something to +// decide about rather than quietly becoming a verdict. +func TestTheAmbiguousPatternsAreLeftUnknownOnPurpose(t *testing.T) { + for _, token := range []string{"etc_passwd", "php_uname"} { + m, known := classify("Exploit", token) + if known { + t.Errorf("%s was mapped to %+v. If that is deliberate, the reasoning written "+ + "beside it in rules.go has to change too — it currently says the opposite", token, m) + } + if m.category != schema.CategoryOther || m.severity != schema.SeverityMedium { + t.Errorf("%s fell back to %q/%q, wanted other/medium", token, m.category, m.severity) + } + } +} + +// The token beats the rule name, and a hash falls through to it. +// +// The whole point of D-057: Function (eval) is a backdoor because of `eval`, not because +// of "Function"; Signature (11413268) is known malware because of "Signature", since no +// table will ever hold that hash. +func TestTheTokenIsTriedBeforeTheRuleName(t *testing.T) { + cases := []struct { + rule, token string + category schema.Category + confidence schema.Confidence + }{ + {"Function", "eval", schema.CategoryBackdoor, schema.ConfidenceHeuristic}, + {"Function", "exec", schema.CategoryWebshell, schema.ConfidenceHeuristic}, + {"Exploit", "execution", schema.CategoryBackdoor, schema.ConfidenceHeuristic}, + {"Signature", "11413268", schema.CategoryKnownMalware, schema.ConfidenceSignature}, + } + for _, tc := range cases { + m, known := classify(tc.rule, tc.token) + if !known { + t.Errorf("%s (%s): not classified", tc.rule, tc.token) + continue + } + if m.category != tc.category { + t.Errorf("%s (%s): category %q, wanted %q", tc.rule, tc.token, m.category, tc.category) + } + if m.confidence != tc.confidence { + t.Errorf("%s (%s): confidence %q, wanted %q", tc.rule, tc.token, m.confidence, tc.confidence) + } + } +} diff --git a/tests/contract/amwscan_exploit_vocabulary_test.go b/tests/contract/amwscan_exploit_vocabulary_test.go new file mode 100644 index 0000000..b7ba9c2 --- /dev/null +++ b/tests/contract/amwscan_exploit_vocabulary_test.go @@ -0,0 +1,59 @@ +package contract_test + +import ( + "strings" + "testing" + + "github.com/thiagoluga/SentinelHost/internal/adapter/amwscan" + "github.com/thiagoluga/SentinelHost/internal/schema" +) + +// The Exploit vocabulary is mapped from what the engine says each pattern means. +// +// AMWScan prints a description under every finding, and those descriptions are quoted in +// the rule table next to each entry so the mapping can be checked against the source +// rather than against somebody's reading of a token name: +// +// execution "RCE (Remote Code Execution) allow remote attackers to execute PHP code" +// hex_char "Hex char is usually used for the obfuscation of malicious code" +// +// Before this, every one of them landed on other/medium/heuristic — 732 findings of +// `execution` alone on the account this fixture came from. +func TestTheExploitVocabularyIsClassified(t *testing.T) { + a := amwscan.New().WithStat(fakeStat) + + rep, err := a.Parse(raw(amwscan.Slug, + fixture(t, "amwscan", "real-0.15.1-function-and-exploit.txt"), schema.StatusCompleted)) + if err != nil { + t.Fatalf("Parse: %v", err) + } + + want := map[string]struct { + category schema.Category + severity schema.Severity + }{ + "execution": {schema.CategoryBackdoor, schema.SeverityCritical}, + "hex_char": {schema.CategoryObfuscation, schema.SeverityMedium}, + } + + seen := map[string]bool{} + for _, f := range rep.Findings { + for token, exp := range want { + if !strings.Contains(f.MatchedContent, "("+token+")") { + continue + } + seen[token] = true + if f.Category != exp.category { + t.Errorf("%s: category %q, wanted %q", token, f.Category, exp.category) + } + if f.Severity != exp.severity { + t.Errorf("%s: severity %q, wanted %q", token, f.Severity, exp.severity) + } + } + } + for token := range want { + if !seen[token] { + t.Errorf("the fixture carries no %q finding; it is the evidence this mapping rests on", token) + } + } +}