Wei-Chun Chang

13 papers A* 2A 1B 1Journal 2Unranked 7
YearRankTypeTitle / Venue / Authors
2021 conf
VTC Fall
Wei-Chun Chang, Yao Chiang, Yi Zhang, Hung-Yu Wei
2020 J jnl
IEEE Trans. Comput. Aided Des. Integr. Circuits Syst.
Wei-Chun Chang, Iris Hui-Ru Jiang
2018 A* conf
ICRA
Wei-Chun Chang, Cheng-Wei Wu, Richard Yi-Chia Tsai, Kate Ching-Ju Lin, Yu-Chee Tseng
2017 J jnl
Multim. Tools Appl.
W. G. C. W. Kumara, Shwu-Huey Yen, Hui-Huang Hsu, Timothy K. Shih, Wei-Chun Chang, Enkhtogtokh Togootogtokh
2017 A* conf
DAC
Wei-Chun Chang, Iris Hui-Ru Jiang, Yen-Ting Yu, Wei-Fang Liu
2009 conf
ACIS-ICIS
Ching-Seh Wu, Wei-Chun Chang, Ishwar K. Sethi
2006 conf
IMECS
Suchen Chiang, Wei-Chun Chang
2006 conf
AMT
Wei-Chun Chang, Wei-Cheng Teo, Cheng-Chang Oh Yang
2005 conf
Web Intelligence
Wei-Chun Chang, Ching-Seh Wu, Chun Chang
2003 conf
IWSOC
Kuo-Hsing Cheng, Wei-Chun Chang, Chia Ming Tu
2003 conf
ISCAS (5)
Kuo-Hsing Cheng, Yang-Han Lee, Wei-Chun Chang
2003 A conf
RE
Alistair G. Sutcliffe, Wei-Chun Chang, Richard Neville
2002 B conf
IEEE Congress on Evolutionary Computation
Alistair G. Sutcliffe, Wei-Chun Chang, Richard Neville
redb/extractors/js_extractors/js_content.py
← Index redb/extractors/js_extractors/js_content.py python
"""Persists raw + normalised text into the generic `code_text_content` table.

Reads the raw source and the deobfuscation result directly from the shared
JSContext so no extra compute happens here — both values are computed once
per sample (the source at JSContext construction, the deobfuscation lazily
on first access) and reused by any extractor that needs them.

`text_normalized` is left NULL when the deobfuscation pass produced no
output, so analysts can distinguish "we tried and got nothing" from
"normalisation succeeded".
"""

import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor


class JSContentExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.content_row = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_CONTENT.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, normalizer_used = self._context.deobfuscated

        self.content_row = {
            "content_type": self._context.content_type,
            "text_raw": src,
            "text_normalized": deobfuscated,  # may be None
            "normalizer_used": normalizer_used,  # may be None
        }
        return self.content_row

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type != "ClickHouseExporter":
            return None
        if not self.content_row:
            return None

        r = self.content_row
        data = [[
            self.sha256,
            r["content_type"],
            r["text_raw"],
            r["text_normalized"],
            r["normalizer_used"],
            datetime.now(timezone.utc),
        ]]

        column_names = [
            "sha256",
            "content_type",
            "text_raw",
            "text_normalized",
            "normalizer_used",
            "analysis_date",
        ]

        column_type_names = [
            "FixedString(64)",
            "LowCardinality(String)",
            "String",
            "Nullable(String)",
            "Nullable(String)",
            "DateTime64(3, 'UTC')",
        ]

        return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_text_content"