Natalia Bushuyeva

25 papers Misc 2Journal 1Unranked 22
YearRankTypeTitle / Venue / Authors
2025 conf
ITPM
Natalia Bushuyeva, Yevhen Lobok, Gleb Murovansky
2025 J jnl
Comput.
Sergiy Bushuyev, Natalia Bushuyeva, Ivan Nekrasov, Igor Chumachenko
2024 conf
ITPM
Natalia Bushuyeva, Victoria Bushuieva, Sergey Bushuyev, Kateryna Piliuhina, Jurii Tykchonovych, Alina Zaprivoda, Oleksandr Chernysh
2024 conf
ITPM
Natalia Bushuyeva, Andrii V. Ivko, Andriy Romanov, Mykola Malaksiano, Vadim Romanuke
2024 conf
DTESI
Sergiy Bushuyev, Natalia Bushuyeva, Oleh Ilin, Svetlana Murzabekova, Maira Khusainova, Rakhmatullo Saidullayev
2023 conf
ProfIT AI
Sergey Bushuyev, Natalia Bushuyeva, Victoria Bushuieva, Denis Bushuiev, Andrii V. Ivko
2023 conf
ITPM
Sergey Bushuyev, Natalia Bushuyeva, Victoria Bushuieva, Denis Bushuiev, Liudmyla Tereikovska
2023 Misc conf
CSIT
Sergey Bushuyev, Natalia Bushuyeva, Denis Bushuiev, Victoria Bushuieva
2023 conf
IDAACS
Sergey Bushuyev, Natalia Bushuyeva, Denis Bushuiev, Victoria Bushuieva
2023 conf
ITPM
Sergiy Bushuyev, Natalia Bushuyeva, Svitlana Onyshchenko, Inna Khodikova, Alla Bondar
2023 conf
EUSPN/ICTH
Sergiy Bushuyev, Denis Bushuiev, Victoria Bushuieva, Natalia Bushuyeva, Svetlana Murzabekova
2023 conf
DTESI (workshops, short papers)
Sergey Bushuyev, Natalia Bushuyeva, Victoria Bushuieva, Denis Bushuiev, Svitlana Onyshchenko
2022 conf
ITPM
Sergey Bushuyev, Natalia Bushuyeva, Svitlana Onyshchenko, Alla Bondar
2022 Misc conf
CSIT
Sergey Bushuyev, Natalia Bushuyeva, Denis Bushuiev, Victoria Bushuieva
2022 conf
DTESI
Sergey Bushuyev, Natalia Bushuyeva, Victoria Bushuieva, Denis Bushuiev
2022 conf
COLINS
Sergiy Bushuyev, Natalia Bushuyeva, Victoria Bushuieva, Denis Bushuiev
2022 conf
ITPM
Sergey Bushuyev, Natalia Bushuyeva, Victoria Bushuieva, Denis Bushuiev
2021 conf
CSIT (2)
Sergey Bushuyev, Victoria Bushuieva, Natalia Bushuyeva, Denis Bushuiev
2021 conf
ITPM
Sergey Bushuyev, Igbal Babayev, Denis Bushuiev, Natalia Bushuyeva, Jahid Babayev
2021 conf
CSIT (2)
Ivan Oberemok, Nataliia Oberemok, Natalia Bushuyeva
2021 conf
CSIT (2)
Sergey Bushuyev, Svitlana Onyshchenko, Natalia Bushuyeva, Alla Bondar
2020 conf
CSIT (2)
Alla Bondar, Natalia Bushuyeva, Sergey Bushuyev, Svitlana Onyshchenko
2020 conf
CSIT (2)
Sergey Bushuyev, Denis Bushuiev, Natalia Bushuyeva, Victoria Bushuieva
2019 conf
CSIT (3)
Natalia Bushuyeva, Maryna Kutsenko
2018 conf
CSIT (2)
Natalia Bushuyeva, Denis Bushuiev, Victoriia Busuieva, Igor Achkasov
redb/extractors/js_extractors/js_deobfuscation.py
← Index redb/extractors/js_extractors/js_deobfuscation.py python
import hashlib
import inspect
import re
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor
from redb.extractors.js_extractors.js_patterns import PATTERNS

# String literals of 4+ characters; only used by the deobfuscation diff to count
# strings revealed after deobfuscation. Compiled once at module load.
_STRING_LITERAL_4PLUS_RE = re.compile(r"[\"\']([^\"\']{4,})[\"\']")


class JSDeobfuscationExtractor(JSExtractor):
    """Compute pre/post-deobfuscation metrics for a JS sample.

    The actual deobfuscation pass (external tool with jsbeautifier fallback)
    lives on `JSContext.deobfuscated` and is cached per sample, so any other
    extractor that needs the deobfuscated text reads the same value without
    re-running the subprocess. Configure the external tool via env vars:
        JS_DEOBFUSCATOR_PATH    Path or name (default: webcrack)
        JS_DEOBFUSCATE_TIMEOUT  Seconds (default: 60)
    """

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.deobfuscation_result = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_DEOBFUSCATION.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, deobfuscator_used = self._context.deobfuscated
        if deobfuscated is None:
            return None

        original_size = len(src)
        original_entropy = self._context.text_entropy
        deobfuscated_size = len(deobfuscated)
        deobfuscated_entropy = self._calculate_text_entropy(deobfuscated)
        size_change_ratio = round(deobfuscated_size / original_size, 4) if original_size else 0.0

        # Strings revealed by deobfuscation: matched literals are extracted from
        # both versions and the set difference is the count of "new" strings.
        original_strings = set(_STRING_LITERAL_4PLUS_RE.findall(src))
        deobfuscated_strings = set(_STRING_LITERAL_4PLUS_RE.findall(deobfuscated))
        new_strings = deobfuscated_strings - original_strings

        # Suspicious APIs revealed by deobfuscation. Both sides of the diff
        # come from JSContext caches: the raw scan is computed once for the
        # whole pipeline; the deobfuscated scan is computed once and reused
        # by JSSuspiciousAPIsExtractor's revealed_by_deobf rows.
        original_apis = {n for n in self._context.scan if n in PATTERNS}
        deobfuscated_apis = set(self._context.scan_deobfuscated)
        new_apis = deobfuscated_apis - original_apis

        deobfuscated_sha256 = hashlib.sha256(deobfuscated.encode('utf-8')).hexdigest()

        self.deobfuscation_result = {
            'deobfuscator_used': deobfuscator_used,
            'deobfuscation_successful': True,
            'original_size': original_size,
            'deobfuscated_size': deobfuscated_size,
            'size_change_ratio': size_change_ratio,
            'original_entropy': original_entropy,
            'deobfuscated_entropy': deobfuscated_entropy,
            'new_strings_found': len(new_strings),
            'new_apis_found': len(new_apis),
            'deobfuscated_sha256': deobfuscated_sha256,
        }
        return self.deobfuscation_result

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type == "ClickHouseExporter":
            if not self.deobfuscation_result:
                return None

            r = self.deobfuscation_result
            current_time = datetime.now(timezone.utc)
            data = [[
                self.sha256,
                r['deobfuscator_used'],
                int(r['deobfuscation_successful']),
                r['original_size'],
                r['deobfuscated_size'],
                r['size_change_ratio'],
                r['original_entropy'],
                r['deobfuscated_entropy'],
                r['new_strings_found'],
                r['new_apis_found'],
                r['deobfuscated_sha256'],
                current_time,
            ]]

            column_names = [
                "sha256", "deobfuscator_used", "deobfuscation_successful",
                "original_size", "deobfuscated_size", "size_change_ratio",
                "original_entropy", "deobfuscated_entropy",
                "new_strings_found", "new_apis_found",
                "deobfuscated_sha256", "analysis_date",
            ]

            column_type_names = [
                "FixedString(64)", "LowCardinality(String)", "UInt8",
                "UInt64", "UInt64", "Float64",
                "Float64", "Float64",
                "UInt32", "UInt32",
                "FixedString(64)", "DateTime64(3, 'UTC')",
            ]

            return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "redb_js_deobfuscation"