Neha Sharma

14 papers B 3Misc 1Journal 4Unranked 6
YearRankTypeTitle / Venue / Authors
2025 B conf
WCNC
Neha Sharma, Manojkumar B. Kokare, Swaminathan Ramabadran, Sumit Gautam
2025 conf
VTC2025-Spring
Neha Sharma, Manojkumar B. Kokare, R. Swaminathan, Sumit Gautam
2025 conf
ICC
Manojkumar B. Kokare, Sumit Gautam, R. Swaminathan, Neha Sharma, Aryan Kaushik, Symeon Chatzinotas
2025 conf
NCC
Manojkumar B. Kokare, Rishikesh Mishra, Neha Sharma, Swaminathan Ramabadran, Sumit Gautam
2025 Misc conf
COMSNETS
Emani N. S. S. Anjana, Neha Sharma, Sumit Gautam
2024 J jnl
IEEE Commun. Lett.
Neha Sharma, Sumit Gautam, Symeon Chatzinotas, Björn E. Ottersten
2024 B conf
GLOBECOM
Neha Sharma, Sumit Gautam, Aryan Kaushik, Symeon Chatzinotas, Björn E. Ottersten
2024 J jnl
IEEE Access
Neha Sharma, Sumit Gautam, Symeon Chatzinotas, Björn E. Ottersten
2024 conf
NCC
Neha Sharma, Sumit Gautam
2023 B conf
GLOBECOM
Neha Sharma, Sumit Gautam, Symeon Chatzinotas, Björn E. Ottersten
2023 J jnl
IEEE Commun. Lett.
Prabhat Kumar Sharma, Neha Sharma, Shivani Dhok, Anamika Singh
2023 conf
ICCCNT
Neha Sharma, Sumit Gautam
2021 J jnl
Int. J. Model. Simul. Sci. Comput.
Neha Sharma, Rajeevan Chandel
2017 conf
IC3
Parmjit Singh, Rajeevan Chandel, Neha Sharma
redb/extractors/js_extractors/js_content.py
← Index redb/extractors/js_extractors/js_content.py python
"""Persists raw + normalised text into the generic `code_text_content` table.

Reads the raw source and the deobfuscation result directly from the shared
JSContext so no extra compute happens here — both values are computed once
per sample (the source at JSContext construction, the deobfuscation lazily
on first access) and reused by any extractor that needs them.

`text_normalized` is left NULL when the deobfuscation pass produced no
output, so analysts can distinguish "we tried and got nothing" from
"normalisation succeeded".
"""

import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor


class JSContentExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.content_row = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_CONTENT.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, normalizer_used = self._context.deobfuscated

        self.content_row = {
            "content_type": self._context.content_type,
            "text_raw": src,
            "text_normalized": deobfuscated,  # may be None
            "normalizer_used": normalizer_used,  # may be None
        }
        return self.content_row

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type != "ClickHouseExporter":
            return None
        if not self.content_row:
            return None

        r = self.content_row
        data = [[
            self.sha256,
            r["content_type"],
            r["text_raw"],
            r["text_normalized"],
            r["normalizer_used"],
            datetime.now(timezone.utc),
        ]]

        column_names = [
            "sha256",
            "content_type",
            "text_raw",
            "text_normalized",
            "normalizer_used",
            "analysis_date",
        ]

        column_type_names = [
            "FixedString(64)",
            "LowCardinality(String)",
            "String",
            "Nullable(String)",
            "Nullable(String)",
            "DateTime64(3, 'UTC')",
        ]

        return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_text_content"