Vigor Yang

11 papers C 1Journal 10
YearRankTypeTitle / Venue / Authors
2023 J jnl
CoRR
Weiming Ding, Haoxiang Huang, Tzu Jung Lee, Yingjie Liu, Vigor Yang
2022 J jnl
CoRR
Haoxiang Huang, Yingjie Liu, Vigor Yang
2021 J jnl
J. Comput. Phys.
Petro Junior Milan, Jean-Pierre Hickey, Xingjian Wang, Vigor Yang
2021 J jnl
CoRR
Haoxiang Huang, Yingjie Liu, Vigor Yang
2019 J jnl
J. Comput. Phys.
Yanxing Wang, Vigor Yang
2019 J jnl
J. Comput. Phys.
Yanxing Wang, Xiaodong Chen, Xingjian Wang, Vigor Yang
2018 J jnl
CoRR
Yu-Hung Chang, Liwei Zhang, Xingjian Wang, Shiang-Ting Yeh, Simon Mak, Chih-Li Sung, C. F. Jeff Wu, Vigor Yang
2017 J jnl
CoRR
Shiang-Ting Yeh, Xingjian Wang, Chih-Li Sung, Simon Mak, Yu-Hung Chang, Liwei Zhang, C. F. Jeff Wu, Vigor Yang
2014 J jnl
J. Comput. Phys.
Xiaodong Chen, Vigor Yang
2006 J jnl
J. Aerosp. Comput. Inf. Commun.
Devendra Kumar Tolani, Murat Yasar, Asok Ray, Vigor Yang
2000 C conf
ACC
Boe-Shong Hong, Asok Ray, Vigor Yang
redb/extractors/js_extractors/js_content.py
← Index redb/extractors/js_extractors/js_content.py python
"""Persists raw + normalised text into the generic `code_text_content` table.

Reads the raw source and the deobfuscation result directly from the shared
JSContext so no extra compute happens here — both values are computed once
per sample (the source at JSContext construction, the deobfuscation lazily
on first access) and reused by any extractor that needs them.

`text_normalized` is left NULL when the deobfuscation pass produced no
output, so analysts can distinguish "we tried and got nothing" from
"normalisation succeeded".
"""

import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor


class JSContentExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.content_row = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_CONTENT.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, normalizer_used = self._context.deobfuscated

        self.content_row = {
            "content_type": self._context.content_type,
            "text_raw": src,
            "text_normalized": deobfuscated,  # may be None
            "normalizer_used": normalizer_used,  # may be None
        }
        return self.content_row

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type != "ClickHouseExporter":
            return None
        if not self.content_row:
            return None

        r = self.content_row
        data = [[
            self.sha256,
            r["content_type"],
            r["text_raw"],
            r["text_normalized"],
            r["normalizer_used"],
            datetime.now(timezone.utc),
        ]]

        column_names = [
            "sha256",
            "content_type",
            "text_raw",
            "text_normalized",
            "normalizer_used",
            "analysis_date",
        ]

        column_type_names = [
            "FixedString(64)",
            "LowCardinality(String)",
            "String",
            "Nullable(String)",
            "Nullable(String)",
            "DateTime64(3, 'UTC')",
        ]

        return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_text_content"