Wei Li

13 papers C 1Journal 10Unranked 2
YearRankTypeTitle / Venue / Authors
2025 conf
UPINLBS
Xiangchen Lu, Ruizhi Chen, Yue Dai, Yuan Wu, Wei Li, Guangyi Guo, Xiaoguang Niu, Liang Chen
2025 J jnl
Inf. Fusion
Yuan Wu, Mengyi He, Wei Li, Izzy Yi Jian, Yue Yu, Liang Chen, Ruizhi Chen
2023 J jnl
Remote. Sens.
Xingyu Zheng, Ruizhi Chen, Liang Chen, Lei Wang, Yue Yu, Zhenbing Zhang, Wei Li, Yu Pei, Dewen Wu, Yanlin Ruan
2023 J jnl
IEEE Internet Things J.
Yuan Wu, Ruizhi Chen, Wenju Fu, Wei Li, Haitao Zhou
2023 J jnl
Geo spatial Inf. Sci.
Yuan Wu, Ruizhi Chen, Wenju Fu, Wei Li, Haitao Zhou, Guangyi Guo
2022 J jnl
IEEE Internet Things J.
Yue Yu, Ruizhi Chen, Liang Chen, Wei Li, Yuan Wu, Haitao Zhou
2021 J jnl
IEEE Internet Things J.
Yue Yu, Ruizhi Chen, Liang Chen, Xingyu Zheng, Dewen Wu, Wei Li, Yuan Wu
2021 J jnl
IEEE Commun. Lett.
Yue Yu, Ruizhi Chen, Liang Chen, Wei Li, Yuan Wu, Haitao Zhou
2020 J jnl
IEEE Internet Things J.
Yue Yu, Ruizhi Chen, Liang Chen, Shihao Xu, Wei Li, Yuan Wu, Haitao Zhou
2018 J jnl
J. Sensors
He Huang, Wei Li, De An Luo, Dongwei Qiu, Yang Gao
2018 J jnl
Comput. Electr. Eng.
He Huang, Dong Ha Lee, Kun Chang, Wei Li, Tri Dev Acharya
2018 conf
UPINLBS
Kaiyue Qi, He Huang, Wei Li, De An Luo
2014 C conf
IGARSS
Hui Liu, Lei Pang, Lei Zhang, Xin Huo, Jinying Luan, Kexin Lan, Changfeng Jing, Wei Li
redb/extractors/js_extractors/js_content.py
← Index redb/extractors/js_extractors/js_content.py python
"""Persists raw + normalised text into the generic `code_text_content` table.

Reads the raw source and the deobfuscation result directly from the shared
JSContext so no extra compute happens here — both values are computed once
per sample (the source at JSContext construction, the deobfuscation lazily
on first access) and reused by any extractor that needs them.

`text_normalized` is left NULL when the deobfuscation pass produced no
output, so analysts can distinguish "we tried and got nothing" from
"normalisation succeeded".
"""

import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor


class JSContentExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.content_row = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_CONTENT.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, normalizer_used = self._context.deobfuscated

        self.content_row = {
            "content_type": self._context.content_type,
            "text_raw": src,
            "text_normalized": deobfuscated,  # may be None
            "normalizer_used": normalizer_used,  # may be None
        }
        return self.content_row

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type != "ClickHouseExporter":
            return None
        if not self.content_row:
            return None

        r = self.content_row
        data = [[
            self.sha256,
            r["content_type"],
            r["text_raw"],
            r["text_normalized"],
            r["normalizer_used"],
            datetime.now(timezone.utc),
        ]]

        column_names = [
            "sha256",
            "content_type",
            "text_raw",
            "text_normalized",
            "normalizer_used",
            "analysis_date",
        ]

        column_type_names = [
            "FixedString(64)",
            "LowCardinality(String)",
            "String",
            "Nullable(String)",
            "Nullable(String)",
            "DateTime64(3, 'UTC')",
        ]

        return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_text_content"