Weigang Li

17 papers A* 1A 1B 2Journal 6Unranked 7
YearRankTypeTitle / Venue / Authors
2024 A conf
EuroSys
Zongpu Zhang, Jiangtao Chen, Banghao Ying, Yahui Cao, Lingyu Liu, Jian Li, Xin Zeng, Junyuan Wang, Weigang Li, Haibing Guan
2024 A* conf
INFOCOM
Shuo Shi, Chao Zhang, Zongpu Zhang, Hubin Zhang, Xin Zeng, Weigang Li, Junyuan Wang, Xiantao Zhang, Yibin Shen, Jian Li, Haibing Guan
2023 J jnl
IEEE Trans. Dependable Secur. Comput.
Zongpu Zhang, Hubin Zhang, Junyuan Wang, Xiaokang Hu, Jian Li, Wenqian Yu, Ping Yu, Weigang Li, Bo Cui, Guodong Zhu, Kapil Sood, Brian Will, Haibing Guan
2022 conf
ICUFN
Kun Qiu, Harry Chang, Ying Wang, Xiahui Yu, Wenjun Zhu, Yingqi Liu, Jianwei Ma, Weigang Li, Xiaobo Liu, Shuo Dai
2022 J jnl
CoRR
Kun Qiu, Harry Chang, Ying Wang, Xiahui Yu, Wenjun Zhu, Yingqi Liu, Jianwei Ma, Weigang Li, Xiaobo Liu, Shuo Dai
2021 conf
ICC
Onur Barut, Yan Luo, Tong Zhang, Weigang Li, Peilong Li
2021 J jnl
CoRR
Onur Barut, Yan Luo, Tong Zhang, Weigang Li, Peilong Li
2021 J jnl
IEEE Trans. Dependable Secur. Comput.
Xiaokang Hu, Jian Li, Changzheng Wei, Weigang Li, Xin Zeng, Ping Yu, Haibing Guan
2020 conf
ICC
Derek Manning, Peilong Li, Xiaoban Wu, Yan Luo, Tong Zhang, Weigang Li
2020 J jnl
CoRR
Onur Barut, Yan Luo, Tong Zhang, Weigang Li, Peilong Li
2020 J jnl
IEEE Trans. Parallel Distributed Syst.
Jian Li, Xiaokang Hu, David Qian, Changzheng Wei, Gordon McFadden, Brian Will, Ping Yu, Weigang Li, Haibing Guan
2019 conf
USENIX ATC
Xiaokang Hu, Fuzong Wang, Weigang Li, Jian Li, Haibing Guan
2018 conf
CNS
Wenqian Yu, Ping Yu, Junyuan Wang, Changzheng Wei, Lu Gong, Weigang Li, Bo Cui, Hari K. Tadepalli, Brian Will
2017 B conf
DCC
Weigang Li
2017 conf
SoCC
Changzheng Wei, Jian Li, Weigang Li, Ping Yu, Haibing Guan
2016 conf
CNS
Wenqian Yu, Weigang Li, Junyuan Wang, Changzheng Wei
2016 B conf
DCC
Weigang Li, Yu Yao
redb/extractors/js_extractors/js_suspicious_apis.py
← Index redb/extractors/js_extractors/js_suspicious_apis.py python
import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor
from redb.extractors.js_extractors.js_patterns import CATEGORIES, PATTERNS


# Backwards-compatible export: `{category: [(raw_pattern_string, api_name), ...]}`
# in canonical PATTERNS insertion order (code_execution, network, filesystem,
# process, registry, crypto_encoding, dom_manipulation). Kept so external
# callers (notably JSDeobfuscationExtractor pre-cleanup) keep working until
# they are migrated to PATTERNS directly.
SUSPICIOUS_APIS: "dict[str, list[tuple[str, str]]]" = {}
for _name, _compiled in PATTERNS.items():
    SUSPICIOUS_APIS.setdefault(CATEGORIES[_name], []).append((_compiled.pattern, _name))


class JSSuspiciousAPIsExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.api_findings = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_SUSPICIOUS_APIS.value

    def _get_context_snippet(self, line, max_len=200):
        """Get a truncated context snippet around a match."""
        line = line.strip()
        if len(line) > max_len:
            return line[:max_len] + "..."
        return line

    def extract(self):
        src = self.js_source
        if not src:
            return None

        # Pass 1: shared per-sample scan over the raw source. The dict contains
        # entries for both PATTERNS and FEATURE_PATTERNS; the loop below only
        # consults PATTERNS keys, so feature-only entries are ignored.
        raw_scan = self._context.scan or {}
        raw_lines = self.lines

        # Pass 2: same patterns over the deobfuscated text, when the
        # deobfuscator produced something meaningfully different. APIs hidden
        # behind one obfuscation layer (Vjw0rm-style array.join + eval,
        # Dean-Edwards packers, jjencode, ...) only surface here. The scan is
        # cached on JSContext so JSDeobfuscationExtractor (which computes the
        # new_apis_found diff) reuses the same result.
        deobf_scan = self._context.scan_deobfuscated
        if deobf_scan:
            deobf_text, _ = self._context.deobfuscated
            deobf_lines = deobf_text.splitlines()
        else:
            deobf_lines = []

        findings = []
        # Iterate PATTERNS in canonical order so output is deterministic and
        # matches the historical category/pattern ordering. For each api_name,
        # raw findings take precedence; if an API is found only in the
        # deobfuscated text, we surface it as a row tagged revealed_by_deobf=1
        # with line numbers / snippets pulled from the deobfuscated source.
        for api_name in PATTERNS:
            raw_info = raw_scan.get(api_name)
            if raw_info:
                line_numbers = raw_info["lines"]
                lines_for_snippets = raw_lines
                revealed_by_deobf = 0
            else:
                deobf_info = deobf_scan.get(api_name)
                if not deobf_info:
                    continue
                line_numbers = deobf_info["lines"]
                lines_for_snippets = deobf_lines
                revealed_by_deobf = 1

            snippets = [
                self._get_context_snippet(lines_for_snippets[ln - 1])
                for ln in line_numbers[:3]
                if 0 < ln <= len(lines_for_snippets)
            ]
            findings.append({
                "api_name": api_name,
                "api_category": CATEGORIES[api_name],
                # Historical semantics: count = number of unique lines with a
                # match, not total in-source match count.
                "call_count": len(line_numbers),
                "line_numbers": line_numbers,
                "context_snippet": " | ".join(snippets),
                "revealed_by_deobf": revealed_by_deobf,
            })

        if not findings:
            return None

        self.api_findings = findings
        return findings

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type == "ClickHouseExporter":
            if not self.api_findings:
                return None

            current_time = datetime.now(timezone.utc)
            data = []
            for f in self.api_findings:
                data.append([
                    self.sha256,
                    f['api_name'],
                    f['api_category'],
                    f['call_count'],
                    f['line_numbers'],
                    f['context_snippet'],
                    f['revealed_by_deobf'],
                    current_time,
                ])

            column_names = [
                "sha256", "api_name", "api_category",
                "call_count", "line_numbers", "context_snippet",
                "revealed_by_deobf",
                "analysis_date",
            ]

            column_type_names = [
                "FixedString(64)", "String", "LowCardinality(String)",
                "UInt32", "Array(UInt32)", "String",
                "UInt8",
                "DateTime64(3, 'UTC')",
            ]

            return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "redb_js_suspicious_apis"