Xia Lv

11 papers A 1Journal 4Unranked 6
YearRankTypeTitle / Venue / Authors
2022 conf
ICIIP
Wenjuan Wang, Shengrong Liang, Xia Lv
2019 A conf
ICME
Zhengning Wang, Longfei Feng, Fanwei Zeng, Guang Hu, Xiang Zhang, Xia Lv, Fengjun Zhang
2019 J jnl
计算机科学
Zhengning Wang, Yang Zhou, Xia Lv, Fanwei Zeng, Xiang Zhang, Fengjun Zhang
2017 J jnl
ISPRS Int. J. Geo Inf.
Liang Wu, Lei Xue, Chaoling Li, Xia Lv, Zhanlong Chen, Baode Jiang, Mingqiang Guo, Zhong Xie
2017 conf
APWeb/WAIM (1)
Xia Lv, Peiquan Jin, Lin Mu, Shouhong Wan, Lihua Yue
2016 J jnl
J. Comb. Optim.
Yuehua Bu, Xia Lv
2016 conf
APWeb (2)
Xia Lv, Peiquan Jin, Lihua Yue
2015 J jnl
Discret. Math. Algorithms Appl.
Yuehua Bu, Xia Lv, Xiaoyan Yan
2013 conf
BIC-TA
Xiaoqiang Song, Xia Lv, Xubo Guo, Zuhai Zheng
2010 conf
GCC
Chaoling Li, Miaomiao Song, Xia Lv, Xiangang Luo, Jianqian Li
2008 conf
ISIP
Shiguang Ju, Zheng Wang, Xia Lv
redb/extractors/js_extractors/js_content.py
← Index redb/extractors/js_extractors/js_content.py python
"""Persists raw + normalised text into the generic `code_text_content` table.

Reads the raw source and the deobfuscation result directly from the shared
JSContext so no extra compute happens here — both values are computed once
per sample (the source at JSContext construction, the deobfuscation lazily
on first access) and reused by any extractor that needs them.

`text_normalized` is left NULL when the deobfuscation pass produced no
output, so analysts can distinguish "we tried and got nothing" from
"normalisation succeeded".
"""

import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor


class JSContentExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.content_row = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_CONTENT.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, normalizer_used = self._context.deobfuscated

        self.content_row = {
            "content_type": self._context.content_type,
            "text_raw": src,
            "text_normalized": deobfuscated,  # may be None
            "normalizer_used": normalizer_used,  # may be None
        }
        return self.content_row

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type != "ClickHouseExporter":
            return None
        if not self.content_row:
            return None

        r = self.content_row
        data = [[
            self.sha256,
            r["content_type"],
            r["text_raw"],
            r["text_normalized"],
            r["normalizer_used"],
            datetime.now(timezone.utc),
        ]]

        column_names = [
            "sha256",
            "content_type",
            "text_raw",
            "text_normalized",
            "normalizer_used",
            "analysis_date",
        ]

        column_type_names = [
            "FixedString(64)",
            "LowCardinality(String)",
            "String",
            "Nullable(String)",
            "Nullable(String)",
            "DateTime64(3, 'UTC')",
        ]

        return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_text_content"