Naijia Liu

14 papers B 3Journal 6Unranked 5
YearRankTypeTitle / Venue / Authors
2025 J jnl
IEEE Access
Emerson Carlos Pedrino, Denis Pereira de Lima, Igor F. Gallon, Valentin O. Roda, Naijia Liu, Gianluca Tempesti
2025 J jnl
IEEE Access
Naijia Liu, Kaira Sekiguchi, Yoshiyuki Nakata, Toshiaki Sugie, Takaaki Yoshino, Yukio Ohsawa
2024 conf
ICSRS
Naijia Liu, Gianluca Tempesti
2024 B conf
IEEE Big Data
Naijia Liu, Kaira Sekiguchi, Yoshiyuki Nakata, Toshiaki Sugie, Takaaki Yoshino, Yukio Ohsawa
2023 conf
CNIOT
Naijia Liu, Yuwen Zhang, Haihua Li, Qizheng Sun
2022 J jnl
IEEE Trans. Instrum. Meas.
Changsheng Liu, Shuxu Liu, Tongchuan Tian, Naijia Liu, Haigen Zhou
2021 conf
AIAM (ACM)
Naijia Liu, Shengdong Nie
2021 J jnl
IEEE Trans. Instrum. Meas.
Changsheng Liu, Chunfeng Zhang, Haigen Zhou, Naijia Liu, Gang Li
2019 J jnl
IEEE Access
Gang Li, Naijia Liu, Chunfeng Zhang, Changsheng Liu
2018 B conf
IWCMC
Tong Qin, Hewu Li, Qian Wu, Naijia Liu, Fenghua Li
2013 conf
ITNG
Qian Wang, Chun Yu, Naijia Liu
2013 B conf
PIMRC
Wan Dong, Xiaofeng Zhong, Naijia Liu, Pengzhi Xu, Jing Wang
2012 J jnl
J. Softw.
Cui Jin, Naijia Liu, Li Qi
2011 conf
IScIDE
Qian Wang, Naijia Liu, ZhiRui Cheng
redb/extractors/js_extractors/js_content.py
← Index redb/extractors/js_extractors/js_content.py python
"""Persists raw + normalised text into the generic `code_text_content` table.

Reads the raw source and the deobfuscation result directly from the shared
JSContext so no extra compute happens here — both values are computed once
per sample (the source at JSContext construction, the deobfuscation lazily
on first access) and reused by any extractor that needs them.

`text_normalized` is left NULL when the deobfuscation pass produced no
output, so analysts can distinguish "we tried and got nothing" from
"normalisation succeeded".
"""

import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor


class JSContentExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.content_row = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_CONTENT.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, normalizer_used = self._context.deobfuscated

        self.content_row = {
            "content_type": self._context.content_type,
            "text_raw": src,
            "text_normalized": deobfuscated,  # may be None
            "normalizer_used": normalizer_used,  # may be None
        }
        return self.content_row

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type != "ClickHouseExporter":
            return None
        if not self.content_row:
            return None

        r = self.content_row
        data = [[
            self.sha256,
            r["content_type"],
            r["text_raw"],
            r["text_normalized"],
            r["normalizer_used"],
            datetime.now(timezone.utc),
        ]]

        column_names = [
            "sha256",
            "content_type",
            "text_raw",
            "text_normalized",
            "normalizer_used",
            "analysis_date",
        ]

        column_type_names = [
            "FixedString(64)",
            "LowCardinality(String)",
            "String",
            "Nullable(String)",
            "Nullable(String)",
            "DateTime64(3, 'UTC')",
        ]

        return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_text_content"