Ran Miao

11 papers A* 1Journal 8Unranked 2
YearRankTypeTitle / Venue / Authors
2026 J jnl
Inf. Fusion
Xueyu Chen, Ran Miao, Liang Hu, Ruijian Wei, Qi Zhang, Kaitao Song, Usman Naseem, Cairong Zhao
2026 J jnl
CoRR
Jichao Zhang, Ran Miao, Limin Li
2025 J jnl
CoRR
Yuan Wei, Xiaohan Shan, Ran Miao, Jianmin Li
2025 conf
ICIC (12)
Yuan Wei, Xiaohan Shan, Ran Miao, Jianmin Li
2024 J jnl
CoRR
Ran Miao, Xueyu Chen, Liang Hu, Zhifei Zhang, Minghua Wan, Qi Zhang, Cairong Zhao
2023 A* conf
ICDM
Ran Miao, Xueyu Chen, Liang Hu, Zhifei Zhang, Minghua Wan, Qi Zhang, Cairong Zhao
2020 J jnl
Medical Biol. Eng. Comput.
Sheng Huang, Feifei Lee, Ran Miao, Qin Si, Chaowen Lu, Qiu Chen
2020 J jnl
J. Vis. Commun. Image Represent.
Qian Zhang, Feifei Lee, Ya-Gang Wang, Ran Miao, Lei Chen, Qiu Chen
2020 J jnl
IEEE Access
Shuai Yang, Feifei Lee, Ran Miao, Jiawei Cai, Lu Chen, Wei Yao, Koji Kotani, Qiu Chen
2020 J jnl
CoRR
Shufan Shen, Ran Miao, Yi Wang, Zhihua Wei
2018 conf
ICIA
Ran Miao
redb/extractors/js_extractors/js_content.py
← Index redb/extractors/js_extractors/js_content.py python
"""Persists raw + normalised text into the generic `code_text_content` table.

Reads the raw source and the deobfuscation result directly from the shared
JSContext so no extra compute happens here — both values are computed once
per sample (the source at JSContext construction, the deobfuscation lazily
on first access) and reused by any extractor that needs them.

`text_normalized` is left NULL when the deobfuscation pass produced no
output, so analysts can distinguish "we tried and got nothing" from
"normalisation succeeded".
"""

import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor


class JSContentExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.content_row = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_CONTENT.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, normalizer_used = self._context.deobfuscated

        self.content_row = {
            "content_type": self._context.content_type,
            "text_raw": src,
            "text_normalized": deobfuscated,  # may be None
            "normalizer_used": normalizer_used,  # may be None
        }
        return self.content_row

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type != "ClickHouseExporter":
            return None
        if not self.content_row:
            return None

        r = self.content_row
        data = [[
            self.sha256,
            r["content_type"],
            r["text_raw"],
            r["text_normalized"],
            r["normalizer_used"],
            datetime.now(timezone.utc),
        ]]

        column_names = [
            "sha256",
            "content_type",
            "text_raw",
            "text_normalized",
            "normalizer_used",
            "analysis_date",
        ]

        column_type_names = [
            "FixedString(64)",
            "LowCardinality(String)",
            "String",
            "Nullable(String)",
            "Nullable(String)",
            "DateTime64(3, 'UTC')",
        ]

        return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_text_content"