Haiying Huang

11 papers Misc 1Journal 5Unranked 5
YearRankTypeTitle / Venue / Authors
2026 J jnl
CoRR
Mengrui Zhang, Bang Huang, Yunxin Xu, Haiying Huang, Luxi Zhao, Mochun Long, Qingyu Song, Qiao Xiang, Xue (Steve) Liu, Jiwu Shu
2024 J jnl
Sensors
Haiying Huang, Wenlu Cai, Yongjian Mao, Kun Wan, Yong Wen, Yuqiang Han, Qiang Zhang, Rong Zhang, Xing Zheng
2024 J jnl
J. Circuits Syst. Comput.
Hua Deng, Haiying Huang, Osama Alfarraj, Amr Tolba
2019 J jnl
IEEE Access
Zhiyang Fang, Junfeng Wang, Boya Li, Siqi Wu, Yingjie Zhou, Haiying Huang
2018 conf
ICPHM
Chiman Kwan, Haiying Huang, Md. Mazharul Islam, Bulent Ayhan
2016 conf
WWW (Companion Volume)
Vladan Radosavljevic, Mihajlo Grbovic, Nemanja Djuric, Narayan Bhamidipati, Daneo Zhang, Jack Wang, Jiankai Dang, Haiying Huang, Ananth Nagarajan, Peiji Chen
2014 conf
CSE
Haiying Huang, Wuyi Zhang, Gaochao Deng, James Chen
2011 conf
ICFCE
Keyin Wang, Chaoyong Guo, Haiying Huang
2011 conf
ICFCE
Haiying Huang, Chaoyong Guo, Keyin Wang
2007 J jnl
Knowl. Inf. Syst.
Jason Van Hulse, Taghi M. Khoshgoftaar, Haiying Huang
2006 Misc conf
PDPTA
Haiying Huang, Jiati Deng, Min Hou
redb/extractors/js_extractors/js_content.py
← Index redb/extractors/js_extractors/js_content.py python
"""Persists raw + normalised text into the generic `code_text_content` table.

Reads the raw source and the deobfuscation result directly from the shared
JSContext so no extra compute happens here — both values are computed once
per sample (the source at JSContext construction, the deobfuscation lazily
on first access) and reused by any extractor that needs them.

`text_normalized` is left NULL when the deobfuscation pass produced no
output, so analysts can distinguish "we tried and got nothing" from
"normalisation succeeded".
"""

import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor


class JSContentExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.content_row = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_CONTENT.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, normalizer_used = self._context.deobfuscated

        self.content_row = {
            "content_type": self._context.content_type,
            "text_raw": src,
            "text_normalized": deobfuscated,  # may be None
            "normalizer_used": normalizer_used,  # may be None
        }
        return self.content_row

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type != "ClickHouseExporter":
            return None
        if not self.content_row:
            return None

        r = self.content_row
        data = [[
            self.sha256,
            r["content_type"],
            r["text_raw"],
            r["text_normalized"],
            r["normalizer_used"],
            datetime.now(timezone.utc),
        ]]

        column_names = [
            "sha256",
            "content_type",
            "text_raw",
            "text_normalized",
            "normalizer_used",
            "analysis_date",
        ]

        column_type_names = [
            "FixedString(64)",
            "LowCardinality(String)",
            "String",
            "Nullable(String)",
            "Nullable(String)",
            "DateTime64(3, 'UTC')",
        ]

        return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_text_content"