Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
54 changes: 54 additions & 0 deletions internal/adapter/amwscan/rules.go
Original file line number Diff line number Diff line change
Expand Up @@ -61,6 +61,60 @@ var ruleTable = map[string]mapping{
"suspicious": {schema.CategoryOther, schema.SeverityMedium, schema.ConfidenceHeuristic},
"unsafe": {schema.CategoryOther, schema.SeverityMedium, schema.ConfidenceHeuristic},
"permissions": {schema.CategorySuspiciousPerms, schema.SeverityMedium, schema.ConfidenceAnomaly},

// AMWScan's Exploit vocabulary, mapped from the descriptions the engine itself prints
// on the "- " line under every finding. Those descriptions are quoted here so the next
// person can check the mapping against the source rather than against my reading of it.
// Captured from a live account (see tests/testdata/raw/amwscan/PROVENANCE.md), where
// `execution` alone accounted for 732 findings.
//
// "RCE (Remote Code Execution) allow remote attackers to execute PHP code on the
// target machine via HTTP" — the same thing `eval` means, which is why it lands in the
// same place.
"execution": {schema.CategoryBackdoor, schema.SeverityCritical, schema.ConfidenceHeuristic},
// "Nano is a family of PHP webshells which are code golfed to be extremely stealthy
// and efficient" — a named webshell family, not a pattern that happens to look bad.
"nano": {schema.CategoryWebshell, schema.SeverityCritical, schema.ConfidenceHeuristic},
// "LFI (Local File Inclusion), through a image inclusion, allow remote attackers to
// inject and execute arbitrary commands or code" — the same category and weight the
// table already gives plain `include`.
"clever_include": {schema.CategoryInjection, schema.SeverityHigh, schema.ConfidenceHeuristic},

// The obfuscation family. Every one of these says "usually used for the obfuscation of
// malicious code", and USUALLY is the operative word: minified libraries, licence
// blobs and legitimate encoders trip them too.
//
// Medium, matching the existing `encoded` entry rather than `obfuscated`. Obfuscation
// on its own is a reason to look, not a reason to act, and a heuristic that fires on
// ordinary vendor code at high severity is one whose severity stops meaning anything.
"base64_long": {schema.CategoryObfuscation, schema.SeverityMedium, schema.ConfidenceHeuristic},
"hex_char": {schema.CategoryObfuscation, schema.SeverityMedium, schema.ConfidenceHeuristic},
"double_var2": {schema.CategoryObfuscation, schema.SeverityMedium, schema.ConfidenceHeuristic},
"concat_vars_array": {schema.CategoryObfuscation, schema.SeverityMedium, schema.ConfidenceHeuristic},
"concat_vars_with_spaces": {schema.CategoryObfuscation, schema.SeverityMedium, schema.ConfidenceHeuristic},

// "Comments composed by 5 random chars usually used to detect if a file is infected
// yet" — a marker malware leaves behind to recognise its own work. Not a technique the
// file performs, so it is not backdoor or webshell; it is a strong sign the file has
// been touched by something that keeps bookkeeping.
"infected_comment": {schema.CategoryOther, schema.SeverityHigh, schema.ConfidenceHeuristic},

// Two are deliberately NOT here, and the reason belongs next to the ones that are.
//
// etc_passwd — "the /etc/passwd file on Unix systems contains password information,
// an attacker who has accessed the etc/passwd file may attempt a brute force
// attack". True of an attacker; also true of every config parser, test fixture and
// tutorial that mentions the path. The engine is describing what the string means
// when an attacker wrote it, and the pattern cannot tell who did.
//
// php_uname — the engine files it under RCE. It is one information-gathering call,
// which diagnostics and installers make legitimately. Calling that critical on its
// own would put ordinary code next to webshells in the same list.
//
// Both stay unknown, which means other/medium/heuristic and a line in the note counts
// so they keep showing up as something to decide about. That is the honest state for a
// pattern whose meaning depends on who wrote the file, and an unmapped rule is not a
// discarded one.
}

// classify translates the engine's rule name into the normalized schema.
Expand Down
85 changes: 85 additions & 0 deletions internal/adapter/amwscan/rules_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,85 @@
package amwscan

import (
"testing"

"github.com/thiagoluga/SentinelHost/internal/schema"
)

// Obfuscation is a reason to look, not a reason to act.
//
// Every entry in that family says the technique is USUALLY used for malicious code, and
// usually is the operative word — minified libraries, licence blobs and legitimate
// encoders trip the same patterns. They sit at medium, matching the existing `encoded`
// entry, because a heuristic that fires on ordinary vendor code at high severity is one
// whose severity stops meaning anything.
func TestTheObfuscationFamilyStaysAtMedium(t *testing.T) {
for _, token := range []string{
"base64_long", "hex_char", "double_var2",
"concat_vars_array", "concat_vars_with_spaces",
} {
m, known := classify("Exploit", token)
if !known {
t.Errorf("%s is not in the table", token)
continue
}
if m.category != schema.CategoryObfuscation {
t.Errorf("%s: category %q, wanted obfuscation", token, m.category)
}
if m.severity != schema.SeverityMedium {
t.Errorf("%s: severity %q, wanted medium — obfuscation alone is not a reason "+
"to act, and inflating it costs the severity field its meaning", token, m.severity)
}
}
}

// The two deliberate omissions stay unknown, and unknown is not discarded.
//
// etc_passwd matches every config parser and tutorial that names the path; php_uname is
// one information-gathering call that installers make legitimately. The engine describes
// what they mean when an attacker wrote them, and the pattern cannot tell who did. They
// come out as other/medium and get counted, so they keep showing up as something to
// decide about rather than quietly becoming a verdict.
func TestTheAmbiguousPatternsAreLeftUnknownOnPurpose(t *testing.T) {
for _, token := range []string{"etc_passwd", "php_uname"} {
m, known := classify("Exploit", token)
if known {
t.Errorf("%s was mapped to %+v. If that is deliberate, the reasoning written "+
"beside it in rules.go has to change too — it currently says the opposite", token, m)
}
if m.category != schema.CategoryOther || m.severity != schema.SeverityMedium {
t.Errorf("%s fell back to %q/%q, wanted other/medium", token, m.category, m.severity)
}
}
}

// The token beats the rule name, and a hash falls through to it.
//
// The whole point of D-057: Function (eval) is a backdoor because of `eval`, not because
// of "Function"; Signature (11413268) is known malware because of "Signature", since no
// table will ever hold that hash.
func TestTheTokenIsTriedBeforeTheRuleName(t *testing.T) {
cases := []struct {
rule, token string
category schema.Category
confidence schema.Confidence
}{
{"Function", "eval", schema.CategoryBackdoor, schema.ConfidenceHeuristic},
{"Function", "exec", schema.CategoryWebshell, schema.ConfidenceHeuristic},
{"Exploit", "execution", schema.CategoryBackdoor, schema.ConfidenceHeuristic},
{"Signature", "11413268", schema.CategoryKnownMalware, schema.ConfidenceSignature},
}
for _, tc := range cases {
m, known := classify(tc.rule, tc.token)
if !known {
t.Errorf("%s (%s): not classified", tc.rule, tc.token)
continue
}
if m.category != tc.category {
t.Errorf("%s (%s): category %q, wanted %q", tc.rule, tc.token, m.category, tc.category)
}
if m.confidence != tc.confidence {
t.Errorf("%s (%s): confidence %q, wanted %q", tc.rule, tc.token, m.confidence, tc.confidence)
}
}
}
59 changes: 59 additions & 0 deletions tests/contract/amwscan_exploit_vocabulary_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,59 @@
package contract_test

import (
"strings"
"testing"

"github.com/thiagoluga/SentinelHost/internal/adapter/amwscan"
"github.com/thiagoluga/SentinelHost/internal/schema"
)

// The Exploit vocabulary is mapped from what the engine says each pattern means.
//
// AMWScan prints a description under every finding, and those descriptions are quoted in
// the rule table next to each entry so the mapping can be checked against the source
// rather than against somebody's reading of a token name:
//
// execution "RCE (Remote Code Execution) allow remote attackers to execute PHP code"
// hex_char "Hex char is usually used for the obfuscation of malicious code"
//
// Before this, every one of them landed on other/medium/heuristic — 732 findings of
// `execution` alone on the account this fixture came from.
func TestTheExploitVocabularyIsClassified(t *testing.T) {
a := amwscan.New().WithStat(fakeStat)

rep, err := a.Parse(raw(amwscan.Slug,
fixture(t, "amwscan", "real-0.15.1-function-and-exploit.txt"), schema.StatusCompleted))
if err != nil {
t.Fatalf("Parse: %v", err)
}

want := map[string]struct {
category schema.Category
severity schema.Severity
}{
"execution": {schema.CategoryBackdoor, schema.SeverityCritical},
"hex_char": {schema.CategoryObfuscation, schema.SeverityMedium},
}

seen := map[string]bool{}
for _, f := range rep.Findings {
for token, exp := range want {
if !strings.Contains(f.MatchedContent, "("+token+")") {
continue
}
seen[token] = true
if f.Category != exp.category {
t.Errorf("%s: category %q, wanted %q", token, f.Category, exp.category)
}
if f.Severity != exp.severity {
t.Errorf("%s: severity %q, wanted %q", token, f.Severity, exp.severity)
}
}
}
for token := range want {
if !seen[token] {
t.Errorf("the fixture carries no %q finding; it is the evidence this mapping rests on", token)
}
}
}
Loading