Haijun Zhai

19 papers C 1Misc 2Journal 7Unranked 9
YearRankTypeTitle / Venue / Authors
2025 J jnl
CoRR
Xinye Tang, Haijun Zhai, Chaitanya Belwal, Vineeth Thayanithi, Philip Baumann, Yogesh K. Roy
2015 J jnl
J. Am. Medical Informatics Assoc.
Yizhao Ni, Stephanie Kennebeck, Judith W. Dexheimer, Constance M. McAneney, Huaxiu Tang, Todd Lingren, Qi Li, Haijun Zhai, Imre Solti
2015 J jnl
J. Biomed. Informatics
Qi Li, Eric S. Kirkendall, Eric S. Hall, Yizhao Ni, Todd Lingren, Megan Kaiser, Nataline Lingren, Haijun Zhai, Imre Solti, Kristin Melton
2014 Misc conf
AMIA
Yizhao Ni, Stephanie Kennebeck, Constance M. McAneney, Judith W. Dexheimer, Todd Lingren, Qi Li, Haijun Zhai, Imre Solti
2014 J jnl
J. Am. Medical Informatics Assoc.
Todd Lingren, Louise Deléger, Katalin Molnár, Haijun Zhai, Jareen Meinzen-Derr, Megan Kaiser, Laura Stoutenborough, Qi Li, Imre Solti
2014 J jnl
J. Am. Medical Informatics Assoc.
Qi Li, Kristin Melton, Todd Lingren, Eric S. Kirkendall, Eric S. Hall, Haijun Zhai, Yizhao Ni, Megan Kaiser, Laura Stoutenborough, Imre Solti
2013 J jnl
J. Am. Medical Informatics Assoc.
Qi Li, Haijun Zhai, Louise Deléger, Todd Lingren, Megan Kaiser, Laura Stoutenborough, Imre Solti
2013 J jnl
BMC Medical Informatics Decis. Mak.
Qi Li, Louise Deléger, Todd Lingren, Haijun Zhai, Megan Kaiser, Laura Stoutenborough, Anil G. Jegga, Kevin Bretonnel Cohen, Imre Solti
2013 conf
BCB
David Solti, Haijun Zhai
2013 Misc conf
AMIA
Haijun Zhai, Patrick Brady, Qi Li, Todd Lingren, Yizhao Ni, Derek S. Wheeler, Imre Solti
2012 conf
HISB
Haijun Zhai, Todd Lingren, Louise Deléger, Qi Li, Megan Kaiser, Laura Stoutenborough, Imre Solti
2012 conf
HISB
Qi Li, Haijun Zhai, Louise Deléger, Todd Lingren, Megan Kaiser, Laura Stoutenborough, Imre Solti
2012 conf
HISB
Todd Lingren, Louise Deléger, Katalin Molnár, Haijun Zhai, Jareen Meinzen-Derr, Megan Kaiser, Laura Stoutenborough, Qi Li, Imre Solti
2012 conf
HISB
Louise Deléger, Holly Brodzinski, Haijun Zhai, Qi Li, Todd Lingren, Eric S. Kirkendall, Evaline Alessandrini, Imre Solti
2012 conf
CGC
Zude Chen, Jianxun Liu, Haijun Zhai, Lei Jiang, Buqing Cao
2010 C conf
ISI
Xiangtao Liu, Xueqi Cheng, Jingyuan Li, Haijun Zhai, Shuo Bai
2009 conf
TREC
Haijun Zhai, Xueqi Cheng, Jiafeng Guo, Hongbo Xu, Yue Liu
2009 conf
Web Intelligence
Haijun Zhai, Jiafeng Guo, Qiong Wu, Xueqi Cheng, Huawei Shen, Jin Zhang
2009 conf
Web Intelligence
Qiong Wu, Songbo Tan, Haijun Zhai, Gang Zhang, Miyi Duan, Xueqi Cheng
redb/extractors/js_extractors/js_suspicious_apis.py
← Index redb/extractors/js_extractors/js_suspicious_apis.py python
import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor
from redb.extractors.js_extractors.js_patterns import CATEGORIES, PATTERNS


# Backwards-compatible export: `{category: [(raw_pattern_string, api_name), ...]}`
# in canonical PATTERNS insertion order (code_execution, network, filesystem,
# process, registry, crypto_encoding, dom_manipulation). Kept so external
# callers (notably JSDeobfuscationExtractor pre-cleanup) keep working until
# they are migrated to PATTERNS directly.
SUSPICIOUS_APIS: "dict[str, list[tuple[str, str]]]" = {}
for _name, _compiled in PATTERNS.items():
    SUSPICIOUS_APIS.setdefault(CATEGORIES[_name], []).append((_compiled.pattern, _name))


class JSSuspiciousAPIsExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.api_findings = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_SUSPICIOUS_APIS.value

    def _get_context_snippet(self, line, max_len=200):
        """Get a truncated context snippet around a match."""
        line = line.strip()
        if len(line) > max_len:
            return line[:max_len] + "..."
        return line

    def extract(self):
        src = self.js_source
        if not src:
            return None

        # Pass 1: shared per-sample scan over the raw source. The dict contains
        # entries for both PATTERNS and FEATURE_PATTERNS; the loop below only
        # consults PATTERNS keys, so feature-only entries are ignored.
        raw_scan = self._context.scan or {}
        raw_lines = self.lines

        # Pass 2: same patterns over the deobfuscated text, when the
        # deobfuscator produced something meaningfully different. APIs hidden
        # behind one obfuscation layer (Vjw0rm-style array.join + eval,
        # Dean-Edwards packers, jjencode, ...) only surface here. The scan is
        # cached on JSContext so JSDeobfuscationExtractor (which computes the
        # new_apis_found diff) reuses the same result.
        deobf_scan = self._context.scan_deobfuscated
        if deobf_scan:
            deobf_text, _ = self._context.deobfuscated
            deobf_lines = deobf_text.splitlines()
        else:
            deobf_lines = []

        findings = []
        # Iterate PATTERNS in canonical order so output is deterministic and
        # matches the historical category/pattern ordering. For each api_name,
        # raw findings take precedence; if an API is found only in the
        # deobfuscated text, we surface it as a row tagged revealed_by_deobf=1
        # with line numbers / snippets pulled from the deobfuscated source.
        for api_name in PATTERNS:
            raw_info = raw_scan.get(api_name)
            if raw_info:
                line_numbers = raw_info["lines"]
                lines_for_snippets = raw_lines
                revealed_by_deobf = 0
            else:
                deobf_info = deobf_scan.get(api_name)
                if not deobf_info:
                    continue
                line_numbers = deobf_info["lines"]
                lines_for_snippets = deobf_lines
                revealed_by_deobf = 1

            snippets = [
                self._get_context_snippet(lines_for_snippets[ln - 1])
                for ln in line_numbers[:3]
                if 0 < ln <= len(lines_for_snippets)
            ]
            findings.append({
                "api_name": api_name,
                "api_category": CATEGORIES[api_name],
                # Historical semantics: count = number of unique lines with a
                # match, not total in-source match count.
                "call_count": len(line_numbers),
                "line_numbers": line_numbers,
                "context_snippet": " | ".join(snippets),
                "revealed_by_deobf": revealed_by_deobf,
            })

        if not findings:
            return None

        self.api_findings = findings
        return findings

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type == "ClickHouseExporter":
            if not self.api_findings:
                return None

            current_time = datetime.now(timezone.utc)
            data = []
            for f in self.api_findings:
                data.append([
                    self.sha256,
                    f['api_name'],
                    f['api_category'],
                    f['call_count'],
                    f['line_numbers'],
                    f['context_snippet'],
                    f['revealed_by_deobf'],
                    current_time,
                ])

            column_names = [
                "sha256", "api_name", "api_category",
                "call_count", "line_numbers", "context_snippet",
                "revealed_by_deobf",
                "analysis_date",
            ]

            column_type_names = [
                "FixedString(64)", "String", "LowCardinality(String)",
                "UInt32", "Array(UInt32)", "String",
                "UInt8",
                "DateTime64(3, 'UTC')",
            ]

            return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "redb_js_suspicious_apis"