Xiaofu Chang

15 papers A* 2A 1B 1C 1Journal 4Unranked 6
YearRankTypeTitle / Venue / Authors
2022 A* conf
KDD
Hui Li, Xing Fu, Ruofan Wu, Jinyu Xu, Kai Xiao, Xiaofu Chang, Weiqiang Wang, Shuai Chen, Leilei Shi, Tao Xiong, Yuan Qi
2021 J jnl
CoRR
Yunfei Chu, Xiaofu Chang, Kunyang Jia, Jingzhen Zhou, Hongxia Yang
2021 J jnl
CoRR
Hui Li, Xing Fu, Ruofan Wu, Jinyu Xu, Kai Xiao, Xiaofu Chang, Weiqiang Wang, Shuai Chen, Leilei Shi, Tao Xiong, Yuan Qi
2021 J jnl
CoRR
Lu Wang, Xiaofu Chang, Shuang Li, Yunfei Chu, Hui Li, Wei Zhang, Xiaofeng He, Le Song, Jingren Zhou, Hongxia Yang
2020 A conf
CIKM
Xiaofu Chang, Xuqin Liu, Jianfeng Wen, Shuang Li, Yanming Fang, Le Song, Yuan Qi
2020 A* conf
ICML
Shuang Li, Lu Wang, Ruizhi Zhang, Xiaofu Chang, Xuqin Liu, Yao Xie, Yuan Qi, Le Song
2013 J jnl
Telecommun. Syst.
Yuan Dong, Gang Qin, Guorui Xiao, Shiguo Lian, Xiaofu Chang
2013 conf
TRECVID
Hongliang Bai, Yuan Dong, Shusheng Cen, Lezi Wang, Lei Liu, Wei Liu, Yunlong Bian, Chong Huang, Nan Zhao, Bo Liu, Yanchao Feng, Peng Li, Xiaofu Chang, Kun Tao
2012 C conf
VCIP
Yuan Dong, Jiwei Zhang, Xiaofu Chang, Jian Zhao
2012 conf
CCIS
Nan Zhao, Yuan Dong, Jiwei Zhang, Xiaofu Chang
2012 conf
TRECVID
Kun Tao, Yuan Dong, Yunlong Bian, Xiaofu Chang, Hongliang Bai, Wei Liu, Feng Zhao, Peng Li, Chengbin Zeng
2011 conf
Multimedia on Mobile Devices / Multimedia Content Access: Algorithms and Systems
Hongliang Bai, Chengyu Dong, Lezi Wang, Gang Qin, Kun Tao, Xiaofu Chang, Yuan Dong
2011 B conf
ICMR
Hongliang Bai, Lezi Wang, Gang Qin, Jiwei Zhang, Kun Tao, Xiaofu Chang, Yuan Dong
2011 conf
TRECVID
Yuan Dong, Kun Tao, Xiaofu Chang, Shan Gao, Jiwei Zhang, Hongliang Bai, Wei Liu, Feng Zhao, Peng Li, Chengbin Zeng
2010 conf
TRECVID
Yuan Dong, Kun Tao, Hongliang Bai, Xiaofu Chang, Chengyu Dong, Jiqing Liu, Shan Gao, Jiwei Zhang, Tianxiang Zhou, Guorui Xiao
redb/extractors/js_extractors/js_content.py
← Index redb/extractors/js_extractors/js_content.py python
"""Persists raw + normalised text into the generic `code_text_content` table.

Reads the raw source and the deobfuscation result directly from the shared
JSContext so no extra compute happens here — both values are computed once
per sample (the source at JSContext construction, the deobfuscation lazily
on first access) and reused by any extractor that needs them.

`text_normalized` is left NULL when the deobfuscation pass produced no
output, so analysts can distinguish "we tried and got nothing" from
"normalisation succeeded".
"""

import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor


class JSContentExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.content_row = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_CONTENT.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, normalizer_used = self._context.deobfuscated

        self.content_row = {
            "content_type": self._context.content_type,
            "text_raw": src,
            "text_normalized": deobfuscated,  # may be None
            "normalizer_used": normalizer_used,  # may be None
        }
        return self.content_row

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type != "ClickHouseExporter":
            return None
        if not self.content_row:
            return None

        r = self.content_row
        data = [[
            self.sha256,
            r["content_type"],
            r["text_raw"],
            r["text_normalized"],
            r["normalizer_used"],
            datetime.now(timezone.utc),
        ]]

        column_names = [
            "sha256",
            "content_type",
            "text_raw",
            "text_normalized",
            "normalizer_used",
            "analysis_date",
        ]

        column_type_names = [
            "FixedString(64)",
            "LowCardinality(String)",
            "String",
            "Nullable(String)",
            "Nullable(String)",
            "DateTime64(3, 'UTC')",
        ]

        return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_text_content"