Kang Deng

19 papers B 4C 2Misc 1Journal 8Unranked 4
YearRankTypeTitle / Venue / Authors
2026 J jnl
Pattern Recognit.
Zhi Lin, Bingwen Wang, Xixi Wang, Yu Zhang, Xiao Wang, Kang Deng, Anjie Peng, Jin Tang, Xing Yang
2026 J jnl
Pattern Recognit.
Kang Deng, Qixiang Chen, Yu Zhang, Zhi Lin, Shenjian Gong, Zhenyu Liang, Anjie Peng, Xing Yang, Defu Lian
2025 J jnl
Inf.
Shuai Dong, Kang Deng, Kun Zou
2023 B conf
ICIP
Hui Zeng, Biwei Chen, Kang Deng, Anjie Peng
2023 conf
ICONIP (12)
Anjie Peng, Kang Deng, Hui Zeng, Kaijun Wu, Wenxin Yu
2022 Misc conf
ICASSP
Hui Zeng, Kang Deng, Biwei Chen, Anjie Peng
2021 J jnl
CoRR
Hui Zeng, Morteza Darvish Morshedi Hosseini, Kang Deng, Anjie Peng, Miroslav Goljan
2021 B conf
ICIP
Kang Deng, Anjie Peng, Wanli Dong, Hui Zeng
2020 conf
ICONIP (2)
Anjie Peng, Kang Deng, Jing Zhang, Shenghai Luo, Hui Zeng, Wenxin Yu
2020 C conf
IWDW
Hui Zeng, Kang Deng, Anjie Peng
2020 J jnl
IEEE Access
Hui Zeng, Yongcai Wan, Kang Deng, Anjie Peng
2015 J jnl
J. Ambient Intell. Humaniz. Comput.
Zhiguang Xiong, Kang Deng, Zhusong Liu, Yanping Liu, Xiaocui Yan
2013 conf
EIDWT
Kang Deng, Zhiguang Xiong, Xiaocui Yan, Yanping Liu
2013 conf
EIDWT
Zhiguang Xiong, Kang Deng, Yanping Liu, Xiaocui Yan
2010 B conf
DaWak
Kang Deng, Osmar R. Zaïane
2010 J jnl
Appl. Math. Comput.
Kang Deng, Zhiguang Xiong
2009 B conf
Discovery Science
Kang Deng, Osmar R. Zaïane
2008 C conf
ICA3PP
Pingpeng Yuan, Hai Jin, Kang Deng, Qingcha Chen
2007 J jnl
Appl. Math. Comput.
Kang Deng, Zhiguang Xiong, Yunqing Huang
redb/extractors/malcontent.py
← Index redb/extractors/malcontent.py python
import inspect
import json
import subprocess
from typing import Any
from datetime import datetime, timezone

from redb.extractors.enum import Tag
from redb.models.dataclasses import Malcontent
from redb.extractors.extractor import Extractor
from dotenv import load_dotenv
import os

load_dotenv(override=True)


class MalcontentExtractor(Extractor):
    """
    Extractor for malcontent tool from chainguard-dev/malcontent.

    Malcontent discovers supply-chain compromises through context, differential
    analysis, and 14,000+ YARA rules. It analyzes binaries and code to detect
    malicious content and suspicious behavioral patterns.

    Binary can be extracted from Docker image:
        docker cp $(docker create cgr.dev/chainguard/malcontent:latest):/usr/bin/mal /usr/local/bin/mal

    Stores full JSON output for materialized view extraction.
    """

    # Cache version at class level to avoid repeated subprocess calls
    _cached_version = None

    def __init__(
        self,
        filepath,
        log,
        exporters=None,
        index_prefix=None,
        elastic_index=None,
        known_benign=False,
        known_malicious=False,
    ):
        super().__init__(
            filepath,
            log,
            exporters,
            index_prefix,
            elastic_index,
            known_benign,
            known_malicious
        )
        self.malcontent = None

    @classmethod
    def _get_malcontent_version(cls, log) -> str:
        """Get malcontent version, cached at class level."""
        if cls._cached_version is not None:
            return cls._cached_version

        malcontent_path = os.getenv("MALCONTENT_PATH", "/usr/local/bin/mal")
        try:
            result = subprocess.run(
                [malcontent_path, "--version"],
                capture_output=True,
                text=True,
                timeout=10
            )
            version_output = result.stdout.strip()
            if result.returncode == 0 and version_output:
                # Parse "malcontent version v1.21.5" -> "1.21.5"
                if version_output.startswith("malcontent version v"):
                    version_output = version_output[len("malcontent version v"):]
                elif version_output.startswith("malcontent version "):
                    version_output = version_output[len("malcontent version "):]
                cls._cached_version = version_output
            else:
                cls._cached_version = "unknown"
        except Exception as e:
            log.warning(f"Could not get malcontent version: {e}")
            cls._cached_version = "unknown"

        return cls._cached_version

    def _extract_malcontent(self):
        self.log.debug(inspect.currentframe().f_code.co_name)
        TIMEOUT = int(os.getenv("MALCONTENT_TIMEOUT", "300"))
        malcontent_path = os.getenv("MALCONTENT_PATH", "/usr/local/bin/mal")

        malcontent_command = [malcontent_path, "analyze", "--format=json", self.filepath]

        import signal

        try:
            process = subprocess.Popen(
                malcontent_command,
                stdout=subprocess.PIPE,
                stderr=subprocess.PIPE,
                text=True,
                preexec_fn=os.setsid
            )

            try:
                stdout, stderr = process.communicate(timeout=TIMEOUT)
                if process.returncode != 0:
                    self.log.error(f"Error running malcontent, return code: {process.returncode}, stderr: {stderr}")
                    return {}
            except subprocess.TimeoutExpired:
                self.log.warning(f"The malcontent command timed out after {TIMEOUT} seconds, terminating process group")
                try:
                    os.killpg(process.pid, signal.SIGTERM)
                    try:
                        process.wait(timeout=3)
                    except subprocess.TimeoutExpired:
                        self.log.warning("Process didn't terminate with SIGTERM, sending SIGKILL")
                        os.killpg(process.pid, signal.SIGKILL)
                    process.wait()
                except (ProcessLookupError, OSError) as e:
                    self.log.warning(f"Error while killing process: {e}")
                return {}

            try:
                malcontent_output = json.loads(stdout)
            except json.JSONDecodeError as e:
                self.log.error(f"Error parsing malcontent output: {e}")
                return {}

            # Unwrap the Files/<path> structure to get the inner content
            # Structure is: {"Files": {"/path/to/file": {<actual content>}}}
            files_dict = malcontent_output.get("Files", {})
            if not files_dict:
                self.log.warning("Malcontent output has no 'Files' key")
                return {}

            # Get the first (and only) file's content
            file_content = next(iter(files_dict.values()), {})
            if not file_content:
                self.log.warning("Malcontent output has empty file content")
                return {}

            # Extract risk score and level from the unwrapped content
            risk_score = file_content.get("RiskScore", 0)
            risk_level = file_content.get("RiskLevel", "")

            version = self._get_malcontent_version(self.log)

            self.malcontent = Malcontent(
                malcontent_dump=json.dumps(file_content),
                version=version,
                risk_score=risk_score,
                risk_level=risk_level
            )
            self.log.debug(f"Malcontent analysis complete, version={version}, risk={risk_level}({risk_score})")

        except Exception as e:
            self.log.error(f"Unexpected error in malcontent extraction: {str(e)}")
            if 'process' in locals() and process.poll() is None:
                try:
                    os.killpg(process.pid, signal.SIGKILL)
                    process.wait()
                except:
                    pass
            return {}

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type == "ClickHouseExporter":
            current_time = datetime.now(timezone.utc)

            data = [[
                self.sha256,
                current_time,
                self.malcontent.version,
                self.malcontent.risk_score,
                self.malcontent.risk_level,
                self.malcontent.malcontent_dump
            ]]

            column_names = [
                'sha256', 'analysis_date',
                'malcontent_version', 'malcontent_risk_score', 'malcontent_risk_level',
                'malcontent_json'
            ]

            column_type_names = [
                'FixedString(64)',
                'DateTime64(3, \'UTC\')',
                'LowCardinality(String)', 'UInt8', 'LowCardinality(String)',
                'JSON'
            ]

            return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "redb_malcontent"

    def extract(self):
        self.log.debug(inspect.currentframe().f_code.co_name)
        try:
            self._extract_malcontent()
            return self.malcontent
        except Exception as e:
            self.log.error(f"Error extracting malcontent: {e}")
            return None

    def tag(self):
        return Tag.MALCONTENT.value