diff --git a/SKILLS.md b/SKILLS.md index 8506ff43e2c..d04840f2599 100644 --- a/SKILLS.md +++ b/SKILLS.md @@ -70,6 +70,9 @@ CAPE (Config And Payload Extraction) is a malware analysis sandbox derived from * **Imports:** Explicit imports only (`from lib import a, b`). No `from lib import *`. Group standard library, 3rd party, and local imports. * **Strings:** Use double quotes (`"`) for strings. (This line was corrected from the original prompt to reflect the actual change needed for the example.) * **Logging:** Use `import logging; log = logging.getLogger(__name__)`. Do not use `print()`. + * Pass arguments lazily, `%`-style: `log.warning("Failed to parse %s: %s", url, err)`. Never pre-format the message (`log.warning(f"...")`, `"..." % x`, `"...".format()`, `+`): formatting is then paid even when the level is disabled, and the varying message breaks log grouping. Enforced by the ruff `G` rules (G001-G004) enabled in `pyproject.toml`. +* **String formatting (non-logging):** Prefer f-strings for building values (`f"0x{n:x}"`, `f"{scheme}://{host}{path}"`) over `%` or `str.format()`. +* **Type hints:** The project requires Python >= 3.10. Use builtin generics and PEP 604 unions (`dict[str, list[int]]`, `str | None`) instead of `typing.Dict`, `typing.List`, `typing.Optional`. `ruff check --target-version py310 --select UP006,UP035,UP045 --fix ` converts touched files. * **Exceptions:** Use custom exceptions from `lib/cuckoo/common/exceptions.py` (e.g., `CuckooOperationalError`). ### Local Development Environment diff --git a/analyzer/windows/lib/core/packages.py b/analyzer/windows/lib/core/packages.py index a3661e05e41..10ba18dcda1 100644 --- a/analyzer/windows/lib/core/packages.py +++ b/analyzer/windows/lib/core/packages.py @@ -149,7 +149,7 @@ def choose_package(file_type, file_name, exports, target): return "archive" elif file_name.endswith(".a3x"): return "autoit" - elif file_name.endswith(("cmd", "bat")) or b"@echo off" in file_content: + elif file_name.endswith(("cmd", "bat")) or b"@echo off" in file_content.lower(): return "batch" elif file_name.endswith(".rdp"): return "rdp" diff --git a/docs/book/src/usage/submit.rst b/docs/book/src/usage/submit.rst index e522e7d5a7a..8548a1f8ae4 100644 --- a/docs/book/src/usage/submit.rst +++ b/docs/book/src/usage/submit.rst @@ -130,6 +130,7 @@ Submission & General - ``dllloader``: Specify a process name to fake the DLL launcher (default is ``rundll32.exe``). - ``pwsh``: For PS1 package, prefer PowerShell Core (``pwsh.exe``) if available. - ``ignore_size_check``: Allow ignoring file size limits (must be enabled in ``conf/web.conf``). +- ``ignore_junk_filter``: Set to ``1`` to force analysis of files that would otherwise be skipped by the junk filter (e.g. ``.txt`` / ``.md`` / ``.yml`` extensions, or names like ``readme`` / ``license``). - ``check_shellcode``: Set to ``0`` to disable shellcode detection during package identification. - ``pre_script_args`` / ``during_script_args``: Command line arguments for pre/during-execution scripts. - ``pre_script_timeout``: Timeout for pre-execution script (default 60s). diff --git a/lib/cuckoo/common/demux.py b/lib/cuckoo/common/demux.py index 84a57fbeafb..bada48299eb 100644 --- a/lib/cuckoo/common/demux.py +++ b/lib/cuckoo/common/demux.py @@ -14,7 +14,7 @@ from lib.cuckoo.common.path_utils import path_exists, path_mkdir, path_write_file from lib.cuckoo.common.quarantine import unquarantine from lib.cuckoo.common.trim_utils import trim_file, trimmed_path -from lib.cuckoo.common.utils import get_options, sanitize_filename +from lib.cuckoo.common.utils import get_options, option_enabled, sanitize_filename sfFile = False try: @@ -296,22 +296,24 @@ def is_valid_package(package: str) -> bool: # ToDo fix return type -def _sf_children(child: Any) -> Tuple[bytes, str, str, int]: +def _sf_children(child: Any, ignore_junk_filter: bool = False) -> Tuple[bytes, str, str, int]: path_to_extract = b"" filename_lower = child.filename.lower() - # Skip junk files - if any(filename_lower.endswith(ext) for ext in JUNK_EXTENSIONS): - return b"", child.platform, child.magic, child.filesize - if any(name in filename_lower for name in JUNK_NAMES): - return b"", child.platform, child.magic, child.filesize + # Skip junk files (unless analysis is forced via ignore_junk_filter option) + if not ignore_junk_filter: + if any(filename_lower.endswith(ext) for ext in JUNK_EXTENSIONS): + return b"", child.platform, child.magic, child.filesize + if any(name in filename_lower for name in JUNK_NAMES): + return b"", child.platform, child.magic, child.filesize if b".github/" in filename_lower or b".git/" in filename_lower: return b"", child.platform, child.magic, child.filesize _, ext = os.path.splitext(child.filename) ext = ext.lower() if ( - ext in demux_extensions_list + ignore_junk_filter + or ext in demux_extensions_list or is_valid_package(child.package) or is_valid_type(child.magic) or (not ext and is_valid_type(child.magic)) @@ -341,6 +343,8 @@ def demux_sflock( if os.path.splitext(filename)[1] == b".bin": return retlist, "", submit_opts + ignore_junk = option_enabled(options, "ignore_junk_filter") + # ToDo need to introduce error msgs here try: platform = "" @@ -390,7 +394,7 @@ def demux_sflock( # If 'unpacked.children' already contained the deep files, this loop might need adjusting based on your specific API. execs = find_payload_to_run(getattr(current_child, "filepaths", [])) if execs: - extracted = _sf_children(current_child) + extracted = _sf_children(current_child, ignore_junk_filter=ignore_junk) path = extracted[0] if path: submit_opts += [f"file={runable}" for runable in execs] @@ -398,7 +402,7 @@ def demux_sflock( else: # It's just a single regular file (e.g., malware.exe inside a zip). # Extract and add to task. - extracted = _sf_children(current_child) + extracted = _sf_children(current_child, ignore_junk_filter=ignore_junk) path = extracted[0] if path: retlist.append(extracted) @@ -407,19 +411,19 @@ def demux_sflock( for sf_child in unpacked.children: if sf_child.to_dict().get("children"): for ch in sf_child.children: - tmp_child = _sf_children(ch) + tmp_child = _sf_children(ch, ignore_junk_filter=ignore_junk) # check if path is not empty if tmp_child and tmp_child[0]: retlist.append(tmp_child) # child is not available, the original file should be put into the list if not retlist: - tmp_child = _sf_children(sf_child) + tmp_child = _sf_children(sf_child, ignore_junk_filter=ignore_junk) # check if path is not empty if tmp_child and tmp_child[0]: retlist.append(tmp_child) else: - tmp_child = _sf_children(sf_child) + tmp_child = _sf_children(sf_child, ignore_junk_filter=ignore_junk) # check if path is not empty if tmp_child and tmp_child[0]: retlist.append(tmp_child) @@ -457,14 +461,15 @@ def demux_sample( If file is a ZIP, extract its included files and return their file paths If file is an email, extracts its attachments and return their file paths (later we'll also extract URLs) """ - # Skip junk files - filename_bytes = filename if isinstance(filename, bytes) else filename.encode() - filename_lower_bytes = filename_bytes.lower() - if any(filename_lower_bytes.endswith(ext) for ext in JUNK_EXTENSIONS) or any( - name in filename_lower_bytes for name in JUNK_NAMES - ): - filename_str = filename.decode(errors="ignore") if isinstance(filename, bytes) else filename - return [], [{"junk_filter": f"File {filename_str} skipped by junk filter"}] + # Skip junk files (unless analysis is forced via ignore_junk_filter option) + if not option_enabled(options, "ignore_junk_filter"): + filename_bytes = filename if isinstance(filename, bytes) else filename.encode() + filename_lower_bytes = filename_bytes.lower() + if any(filename_lower_bytes.endswith(ext) for ext in JUNK_EXTENSIONS) or any( + name in filename_lower_bytes for name in JUNK_NAMES + ): + filename_str = filename.decode(errors="ignore") if isinstance(filename, bytes) else filename + return [], [{"junk_filter": f"File {filename_str} skipped by junk filter"}] # sflock requires filename to be bytes object for Py3 # TODO: Remove after checking all uses of demux_sample use bytes ~TheMythologist diff --git a/lib/cuckoo/common/network_utils.py b/lib/cuckoo/common/network_utils.py index 03e1d1f45c8..65cfb2d1ed2 100644 --- a/lib/cuckoo/common/network_utils.py +++ b/lib/cuckoo/common/network_utils.py @@ -68,6 +68,8 @@ _HEX_HANDLE_RE = re.compile(r"^(?:0x)?([0-9a-fA-F]+)$") +WINHTTP_FLAG_SECURE = 0x00800000 + def _norm_domain(d): if not d or not isinstance(d, str): @@ -302,7 +304,7 @@ def _parse_handle(v): if isinstance(v, int): if v <= 0: return None - return "0x%x" % v + return f"0x{v:x}" with suppress(Exception): s = str(v).strip() if not s: @@ -313,7 +315,7 @@ def _parse_handle(v): n = int(m.group(1), 16) if n <= 0: return None - return "0x%x" % n + return f"0x{n:x}" return None @@ -454,7 +456,7 @@ def winhttp_update_from_call(pstate, api_lc, args_map, ret_handle): if conn.get("server") and req.get("object"): scheme = "https" if conn.get("port") == 443 else "http" - req["url"] = "%s://%s%s" % (scheme, conn["server"], req["object"]) + req["url"] = f"{scheme}://{conn['server']}{req['object']}" return # WinHttpSetOption -> applies to session/connect/request by handle @@ -484,68 +486,71 @@ def winhttp_finalize_sessions(state): procs = (state or {}).get("processes") or {} for _, p in procs.items(): - sessions = (p.get("sessions") or {}) - if not sessions: + sessions = p.get("sessions") or {} + connects = p.get("connects") or {} + if not connects: continue sessions_by_domain = {} sessions_by_domain_keys = defaultdict(set) - for s in sessions.values(): + # Walk connects directly: WinHttpOpen may be missing (hooked late / failed), + # which previously orphaned the connect and dropped all its requests. + for c in connects.values(): + if not isinstance(c, dict): + continue + + s = sessions.get(c.get("session_handle")) or {} ua = s.get("user_agent") or "" access_type = s.get("access_type") or "" proxy_name = s.get("proxy_name") or "" proxy_bypass = s.get("proxy_bypass") or "" - for c in s.get("connections") or []: - if not isinstance(c, dict): - continue + dom = _norm_domain(c.get("server") or "") + if not dom: + continue + + port = c.get("port") - server = c.get("server") or "" - dom = _norm_domain(server) - if not dom: + for r in c.get("requests") or []: + if not isinstance(r, dict): continue - port = c.get("port") - scheme = "https" if port == 443 else "http" - - for r in c.get("requests") or []: - if not isinstance(r, dict): - continue - - obj = r.get("object") or "" - if not isinstance(obj, str): - obj = str(obj) - - obj = obj.strip() - if not obj: - continue - - if not obj.startswith("/"): - obj = "/" + obj - - verb = r.get("verb") or "" - if not isinstance(verb, str): - verb = str(verb) - - verb = verb.strip().upper() or "GET" - request = f"{verb} {obj} \r\nUser-Agent: {ua}\r\nHost: {dom}\r\n" - entry = { - "uri": obj, - "dport": port, - "method": verb, - "protocol": scheme, - "user_agent": ua, - "request": request, - "access_type": access_type, - "proxy_name": proxy_name, - "proxy_bypass": proxy_bypass, - } - - key = (obj, verb, ua, access_type, proxy_name, proxy_bypass) - if key not in sessions_by_domain_keys[dom]: - sessions_by_domain.setdefault(dom, []).append(entry) - sessions_by_domain_keys[dom].add(key) + obj = str(r.get("object") or "").strip() + if not obj: + continue + if not obj.startswith("/"): + obj = "/" + obj + + verb = str(r.get("verb") or "").strip().upper() or "GET" + + flags = _safe_int(r.get("flags")) or 0 + secure = bool(flags & WINHTTP_FLAG_SECURE) or port == 443 + scheme = "https" if secure else "http" + default_port = 443 if secure else 80 + # port 0 == INTERNET_DEFAULT_PORT + dport = default_port if port in (None, 0) else port + netloc = dom if dport == default_port else f"{dom}:{dport}" + url = f"{scheme}://{netloc}{obj}" + + request = f"{verb} {obj} \r\nUser-Agent: {ua}\r\nHost: {netloc}\r\n" + entry = { + "url": url, + "uri": obj, + "dport": dport, + "method": verb, + "protocol": scheme, + "user_agent": ua, + "request": request, + "access_type": access_type, + "proxy_name": proxy_name, + "proxy_bypass": proxy_bypass, + } + + key = (url, verb, ua, access_type, proxy_name, proxy_bypass) + if key not in sessions_by_domain_keys[dom]: + sessions_by_domain.setdefault(dom, []).append(entry) + sessions_by_domain_keys[dom].add(key) if sessions_by_domain: sessions_list = [{"host": dom, "events": evts} for dom, evts in sessions_by_domain.items()] diff --git a/lib/cuckoo/core/data/tasking.py b/lib/cuckoo/core/data/tasking.py index 4dd1f3c49b0..c3c07e9ced1 100644 --- a/lib/cuckoo/core/data/tasking.py +++ b/lib/cuckoo/core/data/tasking.py @@ -140,6 +140,8 @@ def task_visibility_lock(lock_engine, task_id): sandbox_packages = ( "access", "archive", + "autoit", + "batch", "nsis", "cpl", "reg", diff --git a/modules/processing/network.py b/modules/processing/network.py index 062de036d30..f64d3a2e7e1 100644 --- a/modules/processing/network.py +++ b/modules/processing/network.py @@ -22,7 +22,7 @@ from hashlib import md5, sha1, sha256 from itertools import islice from json import loads -from typing import Any, Dict, List, Optional +from typing import Any from urllib.parse import urlparse, urlunparse import cachetools.func @@ -146,7 +146,7 @@ ip_passlist.add(ip) if enabled_network_passlist and network_passlist_file and os.path.isfile(network_passlist_file): - with open(os.path.join(CUCKOO_ROOT, network_passlist_file), "r") as f: + with open(os.path.join(CUCKOO_ROOT, network_passlist_file)) as f: for cidr in set(f.read().splitlines()): if cidr.startswith("#") or len(cidr.strip()) == 0: # comment or empty line @@ -496,13 +496,13 @@ def _add_dns(self, udpdata, ts): ans = {"type": "A"} try: ans["data"] = socket.inet_ntoa(answer.rdata) - except socket.error: + except OSError: continue elif answer.type == dpkt.dns.DNS_AAAA: ans = {"type": "AAAA"} try: ans["data"] = socket.inet_ntop(socket.AF_INET6, answer.rdata) - except (socket.error, ValueError): + except (OSError, ValueError): continue elif answer.type == dpkt.dns.DNS_CNAME: ans = {"type": "CNAME", "data": answer.cname} @@ -767,7 +767,7 @@ def run(self): try: file = open(self.filepath, "rb") - except (IOError, OSError): + except OSError: log.error("Unable to open %s", self.filepath) return self.results @@ -1129,7 +1129,7 @@ def _import_ja3_fprints(self): """ ja3_fprints = {} if path_exists(self.ja3_file): - with open(self.ja3_file, "r") as fpfile: + with open(self.ja3_file) as fpfile: for line in fpfile: try: ja3 = loads(line) @@ -1140,7 +1140,7 @@ def _import_ja3_fprints(self): return ja3_fprints - def _load_network_map(self) -> Dict: + def _load_network_map(self) -> dict: with suppress(Exception): behavior_net_map = self.results.get("behavior", {}).get("network_map") or {} if not behavior_net_map: @@ -1171,7 +1171,7 @@ def _load_network_map(self) -> Dict: return net_map return {} - def _reconstruct_endpoint_map(self, raw_map) -> Dict[tuple, List[Dict]]: + def _reconstruct_endpoint_map(self, raw_map) -> dict[tuple, list[dict]]: """ Convert JSON-friendly "ip:port" keys back to (ip, int(port)) tuples. """ @@ -1193,7 +1193,7 @@ def _reconstruct_endpoint_map(self, raw_map) -> Dict[tuple, List[Dict]]: continue return endpoint_map - def _pick_best(self, candidates: List[Dict]) -> Optional[Dict]: + def _pick_best(self, candidates: list[dict]) -> dict | None: if not candidates: return None @@ -1203,7 +1203,7 @@ def _pick_best(self, candidates: List[Dict]) -> Optional[Dict]: return candidates[0] - def _match_dns_process(self, dns_entry: Dict, dns_intents: Dict, max_skew_seconds: float = 10.0) -> Optional[Dict]: + def _match_dns_process(self, dns_entry: dict, dns_intents: dict, max_skew_seconds: float = 10.0) -> dict | None: """ Match a network.dns entry to the closest behavior DNS intent by: - same domain @@ -1241,7 +1241,7 @@ def _match_dns_process(self, dns_entry: Dict, dns_intents: Dict, max_skew_second return candidates[0].get("process") - def _pcap_first_epoch(self, network: Dict) -> Optional[float]: + def _pcap_first_epoch(self, network: dict) -> float | None: ts = [] for k in ("dns", "http"): for e in network.get(k) or []: @@ -1250,7 +1250,7 @@ def _pcap_first_epoch(self, network: Dict) -> Optional[float]: ts.append(float(v)) return min(ts) if ts else None - def _build_dns_events_rel(self, network: Dict, dns_intents: Dict, max_skew_seconds: float = 10.0) -> List[Dict]: + def _build_dns_events_rel(self, network: dict, dns_intents: dict, max_skew_seconds: float = 10.0) -> list[dict]: """ Returns a list of dns events: [{"t_rel": float, "process": {...}|None, "request": "example.com"}] @@ -1271,7 +1271,7 @@ def _build_dns_events_rel(self, network: Dict, dns_intents: Dict, max_skew_secon out.sort(key=lambda x: x["t_rel"]) return out - def _nearest_dns_process_by_rel_time(self, dns_events_rel: List[Dict], t_rel: Any, max_skew: float = 5.0) -> Optional[Dict]: + def _nearest_dns_process_by_rel_time(self, dns_events_rel: list[dict], t_rel: Any, max_skew: float = 5.0) -> dict | None: if not dns_events_rel or not isinstance(t_rel, (int, float)): return None @@ -1287,7 +1287,7 @@ def _nearest_dns_process_by_rel_time(self, dns_events_rel: List[Dict], t_rel: An return best.get("process") return None - def _set_proc_fields(self, obj: Dict, proc: Optional[Dict]): + def _set_proc_fields(self, obj: dict, proc: dict | None): """ Add process_id/process_name onto an existing network entry. If proc is None, sets them to None (keeps template stable). @@ -1299,7 +1299,7 @@ def _set_proc_fields(self, obj: Dict, proc: Optional[Dict]): obj["process_id"] = None obj["process_name"] = None - def _process_map(self, network: Dict): + def _process_map(self, network: dict): net_map = self._load_network_map() if not network or not net_map: @@ -1425,29 +1425,37 @@ def _merge_behavior_network(self, network): if not sessions: continue - # Use first session entry as representative - s0 = sessions[0] or {} - method = s0.get("method") or "" - dport = s0.get("port") - uri = s0.get("uri") or "/" - protocol = s0.get("protocol") - - entry = { - "host": hnorm, - "dport": dport, - "uri": uri, - "method": method, - "data": s0.get("request"), - "protocol": protocol, - "access_type": s0.get("access_type"), - "proxy_name": s0.get("proxy_name"), - "proxy_bypass": s0.get("proxy_bypass"), - "source": "behavior", - "process_id": p.get("process_id"), - "process_name": p.get("process_name"), - } + seen_urls = set() + for s in sessions: + s = s or {} + path = s.get("uri") or "/" + protocol = s.get("protocol") or "http" + url = s.get("url") or f"{protocol}://{hnorm}{path}" + if url in seen_urls: + continue + seen_urls.add(url) + + dport = s.get("dport") + entry = { + "host": hnorm, + "dport": dport, + "port": dport, + # network.http template renders only `uri`; PCAP entries store the full URL there. + "uri": url, + "path": path, + "method": s.get("method") or "", + "data": s.get("request"), + "protocol": protocol, + "user_agent": s.get("user_agent"), + "access_type": s.get("access_type"), + "proxy_name": s.get("proxy_name"), + "proxy_bypass": s.get("proxy_bypass"), + "source": "behavior", + "process_id": p.get("process_id"), + "process_name": p.get("process_name"), + } + network.setdefault("http", []).append(entry) - network.setdefault("http", []).append(entry) existing_hosts.add(hnorm) # DNS @@ -1652,7 +1660,7 @@ def get_tlsmaster(self): if not path_exists(dump_tls_log): return tlsmaster - with open(dump_tls_log, "r") as f: + with open(dump_tls_log) as f: for entry in f: try: for m in re.finditer( @@ -1863,8 +1871,7 @@ def packets_for_stream(fobj, offset): ts, raw = next(pcapiter) fobj.seek(offset) - for p in next_connection_packets(pcapiter, linktype=pcap.datalink()): - yield p + yield from next_connection_packets(pcapiter, linktype=pcap.datalink()) def check_pcap_file_type(filepath): diff --git a/tests/test_demux.py b/tests/test_demux.py index e24a9664c41..3d50432d7b9 100644 --- a/tests/test_demux.py +++ b/tests/test_demux.py @@ -2,8 +2,10 @@ # This file is part of Cuckoo Sandbox - http://www.cuckoosandbox.org # See the file 'docs/LICENSE' for copying permission. +import os import pathlib import tempfile +from types import SimpleNamespace import pytest from tcr_misc import get_sample @@ -92,3 +94,59 @@ def test_demux_package(self): def test_options2passwd(self): options = "password=foobar" demux.options2passwd(options) + + def test_demux_junk_filter_default_rejects(self, tmp_path): + junk_file = tmp_path / "polyshell.txt" + junk_file.write_text("echo polyshell") + + # String filename + demuxed, errors = demux.demux_sample(filename=str(junk_file), package=None, options="", use_sflock=False) + assert demuxed == [] + assert any("junk_filter" in err for err in errors) + + # Bytes filename + demuxed, errors = demux.demux_sample(filename=str(junk_file).encode(), package=None, options="", use_sflock=False) + assert demuxed == [] + assert any("junk_filter" in err for err in errors) + + def test_demux_junk_filter_explicitly_disabled(self, tmp_path): + junk_file = tmp_path / "readme.txt" + junk_file.write_text("readme instructions") + + demuxed, errors = demux.demux_sample(filename=str(junk_file), package=None, options="ignore_junk_filter=0", use_sflock=False) + assert demuxed == [] + assert any("junk_filter" in err for err in errors) + + def test_demux_junk_filter_ignored_with_option(self, tmp_path): + junk_file = tmp_path / "polyshell.txt" + junk_file.write_text("echo polyshell") + + for opt in ("ignore_junk_filter=1", "ignore_junk_filter=true", "ignore_junk_filter=yes"): + demuxed, errors = demux.demux_sample(filename=str(junk_file), package=None, options=opt, use_sflock=False) + assert len(demuxed) == 1 + assert demuxed[0][0] == str(junk_file).encode() + assert errors == [] + + def test_sf_children_junk_filter(self): + child = SimpleNamespace( + filename=b"polyshell.txt", + contents=b"echo polyshell", + platform="windows", + magic="ASCII text", + filesize=14, + package=None, + ) + + # By default (ignore_junk_filter=False), junk file is skipped + path, platform, magic, filesize = demux._sf_children(child, ignore_junk_filter=False) + assert path == b"" + assert filesize == 14 + + # When ignore_junk_filter=True, junk file is extracted + path, platform, magic, filesize = demux._sf_children(child, ignore_junk_filter=True) + assert path != b"" + assert os.path.exists(path) + assert open(path, "rb").read() == b"echo polyshell" + assert filesize == 14 + if os.path.exists(path): + os.remove(path) diff --git a/web/templates/submission/index.html b/web/templates/submission/index.html index c208669fa4b..052c3646524 100644 --- a/web/templates/submission/index.html +++ b/web/templates/submission/index.html @@ -548,6 +548,10 @@
Advance check_shellcode Disable shellcode check during package ID (check_shellcode=0) + + ignore_junk_filter + Force analysis of files otherwise skipped by the junk filter (e.g. .txt) + function Exported function/ordinal to execute (DLL)