Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions SKILLS.md
Original file line number Diff line number Diff line change
Expand Up @@ -70,6 +70,9 @@ CAPE (Config And Payload Extraction) is a malware analysis sandbox derived from
* **Imports:** Explicit imports only (`from lib import a, b`). No `from lib import *`. Group standard library, 3rd party, and local imports.
* **Strings:** Use double quotes (`"`) for strings. (This line was corrected from the original prompt to reflect the actual change needed for the example.)
* **Logging:** Use `import logging; log = logging.getLogger(__name__)`. Do not use `print()`.
* Pass arguments lazily, `%`-style: `log.warning("Failed to parse %s: %s", url, err)`. Never pre-format the message (`log.warning(f"...")`, `"..." % x`, `"...".format()`, `+`): formatting is then paid even when the level is disabled, and the varying message breaks log grouping. Enforced by the ruff `G` rules (G001-G004) enabled in `pyproject.toml`.
* **String formatting (non-logging):** Prefer f-strings for building values (`f"0x{n:x}"`, `f"{scheme}://{host}{path}"`) over `%` or `str.format()`.
* **Type hints:** The project requires Python >= 3.10. Use builtin generics and PEP 604 unions (`dict[str, list[int]]`, `str | None`) instead of `typing.Dict`, `typing.List`, `typing.Optional`. `ruff check --target-version py310 --select UP006,UP035,UP045 --fix <files>` converts touched files.
* **Exceptions:** Use custom exceptions from `lib/cuckoo/common/exceptions.py` (e.g., `CuckooOperationalError`).

### Local Development Environment
Expand Down
2 changes: 1 addition & 1 deletion analyzer/windows/lib/core/packages.py
Original file line number Diff line number Diff line change
Expand Up @@ -149,7 +149,7 @@ def choose_package(file_type, file_name, exports, target):
return "archive"
elif file_name.endswith(".a3x"):
return "autoit"
elif file_name.endswith(("cmd", "bat")) or b"@echo off" in file_content:
elif file_name.endswith(("cmd", "bat")) or b"@echo off" in file_content.lower():
return "batch"
elif file_name.endswith(".rdp"):
return "rdp"
Expand Down
1 change: 1 addition & 0 deletions docs/book/src/usage/submit.rst
Original file line number Diff line number Diff line change
Expand Up @@ -130,6 +130,7 @@ Submission & General
- ``dllloader``: Specify a process name to fake the DLL launcher (default is ``rundll32.exe``).
- ``pwsh``: For PS1 package, prefer PowerShell Core (``pwsh.exe``) if available.
- ``ignore_size_check``: Allow ignoring file size limits (must be enabled in ``conf/web.conf``).
- ``ignore_junk_filter``: Set to ``1`` to force analysis of files that would otherwise be skipped by the junk filter (e.g. ``.txt`` / ``.md`` / ``.yml`` extensions, or names like ``readme`` / ``license``).
- ``check_shellcode``: Set to ``0`` to disable shellcode detection during package identification.
- ``pre_script_args`` / ``during_script_args``: Command line arguments for pre/during-execution scripts.
- ``pre_script_timeout``: Timeout for pre-execution script (default 60s).
Expand Down
47 changes: 26 additions & 21 deletions lib/cuckoo/common/demux.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,7 +14,7 @@
from lib.cuckoo.common.path_utils import path_exists, path_mkdir, path_write_file
from lib.cuckoo.common.quarantine import unquarantine
from lib.cuckoo.common.trim_utils import trim_file, trimmed_path
from lib.cuckoo.common.utils import get_options, sanitize_filename
from lib.cuckoo.common.utils import get_options, option_enabled, sanitize_filename

sfFile = False
try:
Expand Down Expand Up @@ -296,22 +296,24 @@ def is_valid_package(package: str) -> bool:


# ToDo fix return type
def _sf_children(child: Any) -> Tuple[bytes, str, str, int]:
def _sf_children(child: Any, ignore_junk_filter: bool = False) -> Tuple[bytes, str, str, int]:
path_to_extract = b""
filename_lower = child.filename.lower()

# Skip junk files
if any(filename_lower.endswith(ext) for ext in JUNK_EXTENSIONS):
return b"", child.platform, child.magic, child.filesize
if any(name in filename_lower for name in JUNK_NAMES):
return b"", child.platform, child.magic, child.filesize
# Skip junk files (unless analysis is forced via ignore_junk_filter option)
if not ignore_junk_filter:
if any(filename_lower.endswith(ext) for ext in JUNK_EXTENSIONS):
return b"", child.platform, child.magic, child.filesize
if any(name in filename_lower for name in JUNK_NAMES):
return b"", child.platform, child.magic, child.filesize
if b".github/" in filename_lower or b".git/" in filename_lower:
return b"", child.platform, child.magic, child.filesize

_, ext = os.path.splitext(child.filename)
ext = ext.lower()
if (
ext in demux_extensions_list
ignore_junk_filter
or ext in demux_extensions_list
or is_valid_package(child.package)
or is_valid_type(child.magic)
or (not ext and is_valid_type(child.magic))
Expand Down Expand Up @@ -341,6 +343,8 @@ def demux_sflock(
if os.path.splitext(filename)[1] == b".bin":
return retlist, "", submit_opts

ignore_junk = option_enabled(options, "ignore_junk_filter")

# ToDo need to introduce error msgs here
try:
platform = ""
Expand Down Expand Up @@ -390,15 +394,15 @@ def demux_sflock(
# If 'unpacked.children' already contained the deep files, this loop might need adjusting based on your specific API.
execs = find_payload_to_run(getattr(current_child, "filepaths", []))
if execs:
extracted = _sf_children(current_child)
extracted = _sf_children(current_child, ignore_junk_filter=ignore_junk)
path = extracted[0]
if path:
submit_opts += [f"file={runable}" for runable in execs]
retlist.append(extracted)
else:
# It's just a single regular file (e.g., malware.exe inside a zip).
# Extract and add to task.
extracted = _sf_children(current_child)
extracted = _sf_children(current_child, ignore_junk_filter=ignore_junk)
path = extracted[0]
if path:
retlist.append(extracted)
Expand All @@ -407,19 +411,19 @@ def demux_sflock(
for sf_child in unpacked.children:
if sf_child.to_dict().get("children"):
for ch in sf_child.children:
tmp_child = _sf_children(ch)
tmp_child = _sf_children(ch, ignore_junk_filter=ignore_junk)
# check if path is not empty
if tmp_child and tmp_child[0]:
retlist.append(tmp_child)

# child is not available, the original file should be put into the list
if not retlist:
tmp_child = _sf_children(sf_child)
tmp_child = _sf_children(sf_child, ignore_junk_filter=ignore_junk)
# check if path is not empty
if tmp_child and tmp_child[0]:
retlist.append(tmp_child)
else:
tmp_child = _sf_children(sf_child)
tmp_child = _sf_children(sf_child, ignore_junk_filter=ignore_junk)
# check if path is not empty
if tmp_child and tmp_child[0]:
retlist.append(tmp_child)
Expand Down Expand Up @@ -457,14 +461,15 @@ def demux_sample(
If file is a ZIP, extract its included files and return their file paths
If file is an email, extracts its attachments and return their file paths (later we'll also extract URLs)
"""
# Skip junk files
filename_bytes = filename if isinstance(filename, bytes) else filename.encode()
filename_lower_bytes = filename_bytes.lower()
if any(filename_lower_bytes.endswith(ext) for ext in JUNK_EXTENSIONS) or any(
name in filename_lower_bytes for name in JUNK_NAMES
):
filename_str = filename.decode(errors="ignore") if isinstance(filename, bytes) else filename
return [], [{"junk_filter": f"File {filename_str} skipped by junk filter"}]
# Skip junk files (unless analysis is forced via ignore_junk_filter option)
if not option_enabled(options, "ignore_junk_filter"):
filename_bytes = filename if isinstance(filename, bytes) else filename.encode()
filename_lower_bytes = filename_bytes.lower()
if any(filename_lower_bytes.endswith(ext) for ext in JUNK_EXTENSIONS) or any(
name in filename_lower_bytes for name in JUNK_NAMES
):
filename_str = filename.decode(errors="ignore") if isinstance(filename, bytes) else filename
return [], [{"junk_filter": f"File {filename_str} skipped by junk filter"}]

# sflock requires filename to be bytes object for Py3
# TODO: Remove after checking all uses of demux_sample use bytes ~TheMythologist
Expand Down
109 changes: 57 additions & 52 deletions lib/cuckoo/common/network_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -68,6 +68,8 @@

_HEX_HANDLE_RE = re.compile(r"^(?:0x)?([0-9a-fA-F]+)$")

WINHTTP_FLAG_SECURE = 0x00800000


def _norm_domain(d):
if not d or not isinstance(d, str):
Expand Down Expand Up @@ -302,7 +304,7 @@ def _parse_handle(v):
if isinstance(v, int):
if v <= 0:
return None
return "0x%x" % v
return f"0x{v:x}"
with suppress(Exception):
s = str(v).strip()
if not s:
Expand All @@ -313,7 +315,7 @@ def _parse_handle(v):
n = int(m.group(1), 16)
if n <= 0:
return None
return "0x%x" % n
return f"0x{n:x}"
return None


Expand Down Expand Up @@ -454,7 +456,7 @@ def winhttp_update_from_call(pstate, api_lc, args_map, ret_handle):

if conn.get("server") and req.get("object"):
scheme = "https" if conn.get("port") == 443 else "http"
req["url"] = "%s://%s%s" % (scheme, conn["server"], req["object"])
req["url"] = f"{scheme}://{conn['server']}{req['object']}"
return

# WinHttpSetOption -> applies to session/connect/request by handle
Expand Down Expand Up @@ -484,68 +486,71 @@ def winhttp_finalize_sessions(state):
procs = (state or {}).get("processes") or {}

for _, p in procs.items():
sessions = (p.get("sessions") or {})
if not sessions:
sessions = p.get("sessions") or {}
connects = p.get("connects") or {}
if not connects:
continue

sessions_by_domain = {}
sessions_by_domain_keys = defaultdict(set)

for s in sessions.values():
# Walk connects directly: WinHttpOpen may be missing (hooked late / failed),
# which previously orphaned the connect and dropped all its requests.
for c in connects.values():
if not isinstance(c, dict):
continue

s = sessions.get(c.get("session_handle")) or {}
ua = s.get("user_agent") or ""
access_type = s.get("access_type") or ""
proxy_name = s.get("proxy_name") or ""
proxy_bypass = s.get("proxy_bypass") or ""

for c in s.get("connections") or []:
if not isinstance(c, dict):
continue
dom = _norm_domain(c.get("server") or "")
if not dom:
continue

port = c.get("port")

server = c.get("server") or ""
dom = _norm_domain(server)
if not dom:
for r in c.get("requests") or []:
if not isinstance(r, dict):
continue

port = c.get("port")
scheme = "https" if port == 443 else "http"

for r in c.get("requests") or []:
if not isinstance(r, dict):
continue

obj = r.get("object") or ""
if not isinstance(obj, str):
obj = str(obj)

obj = obj.strip()
if not obj:
continue

if not obj.startswith("/"):
obj = "/" + obj

verb = r.get("verb") or ""
if not isinstance(verb, str):
verb = str(verb)

verb = verb.strip().upper() or "GET"
request = f"{verb} {obj} \r\nUser-Agent: {ua}\r\nHost: {dom}\r\n"
entry = {
"uri": obj,
"dport": port,
"method": verb,
"protocol": scheme,
"user_agent": ua,
"request": request,
"access_type": access_type,
"proxy_name": proxy_name,
"proxy_bypass": proxy_bypass,
}

key = (obj, verb, ua, access_type, proxy_name, proxy_bypass)
if key not in sessions_by_domain_keys[dom]:
sessions_by_domain.setdefault(dom, []).append(entry)
sessions_by_domain_keys[dom].add(key)
obj = str(r.get("object") or "").strip()
if not obj:
continue
if not obj.startswith("/"):
obj = "/" + obj

verb = str(r.get("verb") or "").strip().upper() or "GET"

flags = _safe_int(r.get("flags")) or 0
secure = bool(flags & WINHTTP_FLAG_SECURE) or port == 443
scheme = "https" if secure else "http"
default_port = 443 if secure else 80
# port 0 == INTERNET_DEFAULT_PORT
dport = default_port if port in (None, 0) else port
netloc = dom if dport == default_port else f"{dom}:{dport}"
url = f"{scheme}://{netloc}{obj}"

request = f"{verb} {obj} \r\nUser-Agent: {ua}\r\nHost: {netloc}\r\n"
entry = {
"url": url,
"uri": obj,
"dport": dport,
"method": verb,
"protocol": scheme,
"user_agent": ua,
"request": request,
"access_type": access_type,
"proxy_name": proxy_name,
"proxy_bypass": proxy_bypass,
}

key = (url, verb, ua, access_type, proxy_name, proxy_bypass)
if key not in sessions_by_domain_keys[dom]:
sessions_by_domain.setdefault(dom, []).append(entry)
sessions_by_domain_keys[dom].add(key)

if sessions_by_domain:
sessions_list = [{"host": dom, "events": evts} for dom, evts in sessions_by_domain.items()]
Expand Down
2 changes: 2 additions & 0 deletions lib/cuckoo/core/data/tasking.py
Original file line number Diff line number Diff line change
Expand Up @@ -140,6 +140,8 @@ def task_visibility_lock(lock_engine, task_id):
sandbox_packages = (
"access",
"archive",
"autoit",
"batch",
"nsis",
"cpl",
"reg",
Expand Down
Loading
Loading