Xiaofan Zhu

19 papers Misc 2Journal 17
YearRankTypeTitle / Venue / Authors
2023 J jnl
Sensors
Xiaofan Zhu, Pingping Huang, Wei Xu, Weixian Tan, Yaolong Qi
2023 J jnl
BMC Bioinform.
Nan Shao, Chenshuo Ren, Tianyuan Hu, Dianbing Wang, Xiaofan Zhu, Min Li, Tao Cheng, Yingchi Zhang, Xian-En Zhang
2023 J jnl
Int. J. Digit. Earth
Wenhao Liu, Ren Li, Tonghua Wu, Xiaoxi Shi, Xiaodong Wu, Lin Zhao, Guojie Hu, Jimin Yao, Junjie Ma, Shenning Wang, Yao Xiao, Xiaofan Zhu, Jianzong Shi, Dong Wang, Yongping Qiao
2023 J jnl
Int. J. Appl. Earth Obs. Geoinformation
Peiqing Lou, Tonghua Wu, Jie Chen, Bolin Fu, Xiaofan Zhu, Jianjun Chen, Xiaodong Wu, Sizhong Yang, Ren Li, Xingchen Lin, Chengpeng Shang, Amin Wen, Dong Wang, Yune La, Xin Ma
2023 J jnl
Remote. Sens.
Wenhao Liu, Ren Li, Tonghua Wu, Xiaoqian Shi, Lin Zhao, Xiaodong Wu, Guojie Hu, Jimin Yao, Dong Wang, Yao Xiao, Junjie Ma, Yongliang Jiao, Shenning Wang, Defu Zou, Xiaofan Zhu, Jie Chen, Jianzong Shi, Yongping Qiao
2022 J jnl
Remote. Sens.
Jie Chen, Jing Zhang, Tonghua Wu, Junming Hao, Xiaodong Wu, Xuyan Ma, Xiaofan Zhu, Peiqing Lou, Lina Zhang
2022 J jnl
Remote. Sens.
Chengpeng Shang, Tonghua Wu, Ning Ma, Jiemin Wang, Xiangfei Li, Xiaofan Zhu, Tianye Wang, Guojie Hu, Ren Li, Sizhong Yang, Jie Chen, Jimin Yao, Cheng Yang
2022 J jnl
Nat. Comput. Sci.
Xueou Liu, Yigeng Cao, Ye Guo, Xiaowen Gong, Yahui Feng, Yao Wang, Mingyang Wang, Mengxuan Cui, Wenwen Guo, Luyang Zhang, Ningning Zhao, Xiaoqiang Song, Xuetong Zheng, Xia Chen, Qiujin Shen, Song Zhang, Zhen Song, Linfeng Li, Sizhou Feng, Mingzhe Han, Xiaofan Zhu, Erlie Jiang, Junren Chen
2022 J jnl
Nat. Comput. Sci.
Xueou Liu, Yigeng Cao, Ye Guo, Xiaowen Gong, Yahui Feng, Yao Wang, Mingyang Wang, Mengxuan Cui, Wenwen Guo, Luyang Zhang, Ningning Zhao, Xiaoqiang Song, Xuetong Zheng, Xia Chen, Qiujin Shen, Song Zhang, Zhen Song, Linfeng Li, Sizhou Feng, Mingzhe Han, Xiaofan Zhu, Erlie Jiang, Junren Chen
2021 J jnl
Sensors
Xingyue Zhu, Kaixiong Yu, Xiaofan Zhu, Juan Su, Chi Wu
2021 J jnl
IEEE Access
Xingyue Zhu, Kaixiong Yu, Xiaofan Zhu, Chi Wu
2021 J jnl
Earth Sci. Informatics
Lei Gao, Yazhou Zhou, Kairui Guo, Yong Huang, Xiaofan Zhu
2020 J jnl
Remote. Sens.
Cheng Yang, Tonghua Wu, Jimin Yao, Ren Li, Changwei Xie, Guojie Hu, Xiaofan Zhu, Yinghui Zhang, Jie Ni, Junming Hao, Xiangfei Li, Wensi Ma, Amin Wen
2019 J jnl
CoRR
Sina Masnadi, Joseph J. LaViola Jr., Jana Pavlasek, Xiaofan Zhu, Karthik Desingh, Odest Chadwicke Jenkins
2019 J jnl
Remote. Sens.
Cheng Yang, Tonghua Wu, Jiemin Wang, Jimin Yao, Ren Li, Lin Zhao, Changwei Xie, Xiaofan Zhu, Jie Ni, Junming Hao
2019 J jnl
Remote. Sens.
Junming Hao, Tonghua Wu, Xiaodong Wu, Guojie Hu, Defu Zou, Xiaofan Zhu, Lin Zhao, Ren Li, Changwei Xie, Jie Ni, Cheng Yang, Xiangfei Li, Wensi Ma
2017 J jnl
Genom. Proteom. Bioinform.
Nan Ding, Zhaojun Zhang, Wenyu Yang, Lan Ren, Yingchi Zhang, Jingliao Zhang, Zhanqi Li, Peihong Zhang, Xiaofan Zhu, Xiaojuan Chen, Xiangdong Fang
2012 Misc conf
ICASSP
Xiaofan Zhu, Michael G. Rabbat
2012 Misc conf
ICASSP
Xiaofan Zhu, Michael G. Rabbat
redb/extractors/js_extractors/js_strings.py
← Index redb/extractors/js_extractors/js_strings.py python
import base64
import bisect
import inspect
import re
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor
from redb.extractors.js_extractors.js_patterns import STRING_PATTERNS, line_offsets

# Local aliases for the compiled patterns this extractor uses. Defined and
# compiled exactly once in js_patterns.STRING_PATTERNS.
_HEX_STRING_RE = STRING_PATTERNS["hex_escape_seq"]
_UNICODE_STRING_RE = STRING_PATTERNS["unicode_escape_seq"]
_CHARCODE_RE = STRING_PATTERNS["charcode_call"]
_BASE64_STRING_RE = STRING_PATTERNS["base64_quoted"]
_CONCAT_STRING_RE = STRING_PATTERNS["concat_chain"]

# Tokeniser used inside _reconstruct_concat to pull each quoted part out of a
# matched concat chain. Compiled once at module load (was recompiled on every
# concat match before).
_CONCAT_TOKEN_RE = re.compile(r'["\']([^"\']*)["\']')


class JSStringsExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.string_findings = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_STRINGS.value

    def _decode_hex_string(self, hex_str):
        """Decode \\x41\\x42 style hex strings."""
        try:
            # Remove \\x prefix and decode
            clean = hex_str.replace('\\x', '')
            return bytes.fromhex(clean).decode('utf-8', errors='replace')
        except Exception:
            return None

    def _decode_unicode_string(self, uni_str):
        """Decode \\u0041\\u0042 style unicode strings."""
        try:
            return uni_str.encode('utf-8').decode('unicode_escape')
        except Exception:
            return None

    def _decode_charcode(self, charcode_str):
        """Decode String.fromCharCode(72, 101, 108, ...) sequences."""
        try:
            codes = [int(c.strip()) for c in charcode_str.split(',') if c.strip().isdigit()]
            return ''.join(chr(c) for c in codes if 0 <= c <= 0x10FFFF)
        except Exception:
            return None

    def _decode_base64(self, b64_str):
        """Attempt to decode base64 string."""
        try:
            decoded = base64.b64decode(b64_str)
            # Check if result is printable text
            text = decoded.decode('utf-8', errors='strict')
            # Only return if it looks like text (>80% printable)
            printable = sum(1 for c in text if c.isprintable() or c in '\n\r\t')
            if printable / len(text) > 0.8:
                return text
        except Exception:
            pass
        return None

    def _reconstruct_concat(self, concat_match):
        """Reconstruct concatenated string parts."""
        try:
            parts = _CONCAT_TOKEN_RE.findall(concat_match)
            return ''.join(parts)
        except Exception:
            return None

    def _find_line_number(self, match_start):
        """1-indexed line number for `match_start`, looked up in O(log L) via
        bisect over `self._line_offsets` (built once per extract() call).

        Replaces the historical `self.js_source[:match_start].count('\\n') + 1`
        which was O(N) per call and quadratic across all matches in a sample.
        """
        return bisect.bisect_right(self._line_offsets, match_start)

    def _scan_text(self, text):
        """Run every encoded-string pattern over `text` and return a list of
        finding dicts. Stateless apart from the per-call `_line_offsets` cache,
        which `_find_line_number` reads — callers must reset it before invoking
        this so line numbers reference the text being scanned, not the previous
        one.
        """
        findings = []

        # Hex-encoded strings
        for m in _HEX_STRING_RE.finditer(text):
            raw = m.group()
            decoded = self._decode_hex_string(raw)
            if decoded and len(decoded) >= 4:
                findings.append({
                    'string': decoded[:4000],
                    'string_raw': raw[:4000],
                    'string_encoding': 'hex',
                    'string_offset': self._find_line_number(m.start()),
                    'string_length': len(decoded),
                    'string_raw_length': len(raw),
                    'string_entropy': self._calculate_text_entropy(decoded),
                })

        # Unicode-encoded strings
        for m in _UNICODE_STRING_RE.finditer(text):
            raw = m.group()
            decoded = self._decode_unicode_string(raw)
            if decoded and len(decoded) >= 3:
                findings.append({
                    'string': decoded[:4000],
                    'string_raw': raw[:4000],
                    'string_encoding': 'unicode',
                    'string_offset': self._find_line_number(m.start()),
                    'string_length': len(decoded),
                    'string_raw_length': len(raw),
                    'string_entropy': self._calculate_text_entropy(decoded),
                })

        # String.fromCharCode sequences
        for m in _CHARCODE_RE.finditer(text):
            raw = m.group()
            decoded = self._decode_charcode(m.group(1))
            if decoded and len(decoded) >= 4:
                findings.append({
                    'string': decoded[:4000],
                    'string_raw': raw[:4000],
                    'string_encoding': 'charcode',
                    'string_offset': self._find_line_number(m.start()),
                    'string_length': len(decoded),
                    'string_raw_length': len(raw),
                    'string_entropy': self._calculate_text_entropy(decoded),
                })

        # Base64-encoded strings
        for m in _BASE64_STRING_RE.finditer(text):
            raw = m.group(0)
            b64_val = m.group(1)
            decoded = self._decode_base64(b64_val)
            if decoded and len(decoded) >= 10:
                findings.append({
                    'string': decoded[:4000],
                    'string_raw': raw[:4000],
                    'string_encoding': 'base64',
                    'string_offset': self._find_line_number(m.start()),
                    'string_length': len(decoded),
                    'string_raw_length': len(raw),
                    'string_entropy': self._calculate_text_entropy(decoded),
                })

        # Concatenated strings (reassembled)
        for m in _CONCAT_STRING_RE.finditer(text):
            raw = m.group()
            reconstructed = self._reconstruct_concat(raw)
            if reconstructed and len(reconstructed) >= 20:
                findings.append({
                    'string': reconstructed[:4000],
                    'string_raw': raw[:4000],
                    'string_encoding': 'concat',
                    'string_offset': self._find_line_number(m.start()),
                    'string_length': len(reconstructed),
                    'string_raw_length': len(raw),
                    'string_entropy': self._calculate_text_entropy(reconstructed),
                })

        return findings

    def extract(self):
        src = self.js_source
        if not src:
            return None

        # Pass 1: raw source. _line_offsets is keyed off whichever text is
        # currently being scanned so _find_line_number resolves to that text.
        self._line_offsets = line_offsets(src)
        findings = self._scan_text(src)

        # Pass 2: deobfuscated text, when the deobfuscator produced something
        # meaningfully different. Same patterns, but a different surface — for
        # samples where the encoded payload is hidden behind an outer wrapper
        # (e.g. array.join() + eval in Vjw0rm/WSH-RAT) only this pass yields
        # any rows at all.
        deobf_text, _ = self._context.deobfuscated
        if deobf_text and deobf_text != src:
            self._line_offsets = line_offsets(deobf_text)
            findings.extend(self._scan_text(deobf_text))

        if not findings:
            return None

        # Deduplicate by decoded string value (raw pass wins on collision: it
        # comes first in `findings`). A string that surfaces only in the
        # deobfuscated text still gets persisted, which is the whole point of
        # the second pass.
        seen_values = set()
        deduped = []
        for f in findings:
            val_key = f['string'][:100]
            if val_key not in seen_values:
                seen_values.add(val_key)
                deduped.append(f)

        self.string_findings = deduped[:500]  # Limit per file
        # Publish to the shared context so post-loop consumers (notably the IOC
        # plumbing in workers.py) can scrape the decoded strings without
        # holding a reference to this extractor instance.
        self._context.decoded_strings = self.string_findings
        return self.string_findings

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type == "ClickHouseExporter":
            if not self.string_findings:
                return None

            data = []
            for f in self.string_findings:
                data.append([
                    self.sha256,
                    f['string'],
                    f['string_raw'],
                    f['string_encoding'],
                    f['string_offset'],
                    f['string_length'],
                    f['string_raw_length'],
                    f['string_entropy'],
                ])

            column_names = [
                "sha256",
                "string",
                "string_raw",
                "string_encoding",
                "string_offset",
                "string_length",
                "string_raw_length",
                "string_entropy",
            ]

            column_type_names = [
                "FixedString(64)",
                "String",
                "String",
                "LowCardinality(String)",
                "UInt64",
                "UInt32",
                "UInt32",
                "Float32",
            ]

            return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_binja_strings_raw"