Xaveer J. M. Leijtens

17 papers Misc 2Unranked 15
YearRankTypeTitle / Venue / Authors
2024 conf
ICTON
Ruud Jansen, Dzmitry Pustakhod, Bart Combee, Xaveer J. M. Leijtens, Kevin A. Williams, Sylwester Latkowski
2023 Misc conf
PSC
Alessio Miranda, Pável Goor, Kevin A. Williams, Xaveer J. M. Leijtens
2023 Misc conf
PSC
Wenjing Tian, Kevin A. Williams, Ronald Dekker, Joost van Kerkhof, Lucas Beste, Xaveer J. M. Leijtens
2023 conf
ICTON
Sylwester Latkowski, Dzmitry Pustakhod, Ruud Jansen, Xaveer J. M. Leijtens, Kevin A. Williams
2020 conf
ICTON
Sylwester Latkowski, Dzmitry Pustakhod, Michail Chatzimichailidis, Xaveer J. M. Leijtens, Kevin A. Williams
2019 conf
OECC/PSC
Marija Trajkovic, Kaoutar Benyahya, Christian Simonneau, Fabrice Blache, Helene Debregeas, Jean-Guy Provost, Kevin A. Williams, Xaveer J. M. Leijtens
2019 conf
ICTON
Sylwester Latkowski, Dzmitry Pustakhod, Michail Chatzimichailidis, Xaveer J. M. Leijtens, Kevin A. Williams
2018 conf
ECOC
Marija Trajkovic, Fabrice Blache, Filipe Jorge, Karim Mekhazni, Jean-Guy Provost, Helene Debregeas, E. den Haan, Luc M. Augustin, Kevin A. Williams, Xaveer J. M. Leijtens
2018 conf
ECOC
X. Zhang, M. Zhao, Y. Lei, Kevin A. Williams, Xaveer J. M. Leijtens, Yuqing Jiao, S. Huang, Zizheng Cao, Antonius M. J. Koonen
2018 conf
ECOC
D. Zhao, Stefanos Andreou, Weiming Yao, Kevin A. Williams, Xaveer J. M. Leijtens
2016 conf
OFC
Zizheng Cao, Qing Wang, Netsanet M. Tessema, Xaveer J. M. Leijtens, F. M. Soares, Antonius M. J. Koonen
2014 conf
OFC
Valentina Moskalenko, Sylwester Latkowski, Tjibbe de Vries, Luc M. Augustin, Xaveer J. M. Leijtens, Meint K. Smit, E. A. J. M. Bente
2014 conf
OFC
Katarzyna Lawniczuk, Christophe Kazmierski, Mike J. Wale, Pawel Szczepanski, Ryszard Piramidowicz, Meint K. Smit, Xaveer J. M. Leijtens
2014 conf
ICTON
Mulham Khoder, Romain Modeste Nguimdo, Xaveer J. M. Leijtens, Jeroen Bolk, Jan Danckaert, Guy Verschaffelt
2013 conf
OFC/NFOEC
Stanislaw Stopinski, Michal Malinowski, Ryszard Piramidowicz, Meint K. Smit, Xaveer J. M. Leijtens
2013 conf
OFC/NFOEC
Katarzyna Lawniczuk, Mike J. Wale, Pawel Szczepanski, Ryszard Piramidowicz, Meint K. Smit, Xaveer J. M. Leijtens
2009 conf
Hot Interconnects
Aaron Albores-Mejia, Kevin A. Williams, Fausto Gomez-Agis, Shangjian Zhang, Harm J. S. Dorren, Xaveer J. M. Leijtens, Tjibbe de Vries, Yok-Siang Oei, Martijn J. R. Heck, Luc M. Augustin, Richard Nötzel, David J. Robbins, Meint K. Smit
redb/extractors/js_extractors/js_deobfuscation.py
← Index redb/extractors/js_extractors/js_deobfuscation.py python
import hashlib
import inspect
import re
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor
from redb.extractors.js_extractors.js_patterns import PATTERNS

# String literals of 4+ characters; only used by the deobfuscation diff to count
# strings revealed after deobfuscation. Compiled once at module load.
_STRING_LITERAL_4PLUS_RE = re.compile(r"[\"\']([^\"\']{4,})[\"\']")


class JSDeobfuscationExtractor(JSExtractor):
    """Compute pre/post-deobfuscation metrics for a JS sample.

    The actual deobfuscation pass (external tool with jsbeautifier fallback)
    lives on `JSContext.deobfuscated` and is cached per sample, so any other
    extractor that needs the deobfuscated text reads the same value without
    re-running the subprocess. Configure the external tool via env vars:
        JS_DEOBFUSCATOR_PATH    Path or name (default: webcrack)
        JS_DEOBFUSCATE_TIMEOUT  Seconds (default: 60)
    """

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.deobfuscation_result = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_DEOBFUSCATION.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, deobfuscator_used = self._context.deobfuscated
        if deobfuscated is None:
            return None

        original_size = len(src)
        original_entropy = self._context.text_entropy
        deobfuscated_size = len(deobfuscated)
        deobfuscated_entropy = self._calculate_text_entropy(deobfuscated)
        size_change_ratio = round(deobfuscated_size / original_size, 4) if original_size else 0.0

        # Strings revealed by deobfuscation: matched literals are extracted from
        # both versions and the set difference is the count of "new" strings.
        original_strings = set(_STRING_LITERAL_4PLUS_RE.findall(src))
        deobfuscated_strings = set(_STRING_LITERAL_4PLUS_RE.findall(deobfuscated))
        new_strings = deobfuscated_strings - original_strings

        # Suspicious APIs revealed by deobfuscation. Both sides of the diff
        # come from JSContext caches: the raw scan is computed once for the
        # whole pipeline; the deobfuscated scan is computed once and reused
        # by JSSuspiciousAPIsExtractor's revealed_by_deobf rows.
        original_apis = {n for n in self._context.scan if n in PATTERNS}
        deobfuscated_apis = set(self._context.scan_deobfuscated)
        new_apis = deobfuscated_apis - original_apis

        deobfuscated_sha256 = hashlib.sha256(deobfuscated.encode('utf-8')).hexdigest()

        self.deobfuscation_result = {
            'deobfuscator_used': deobfuscator_used,
            'deobfuscation_successful': True,
            'original_size': original_size,
            'deobfuscated_size': deobfuscated_size,
            'size_change_ratio': size_change_ratio,
            'original_entropy': original_entropy,
            'deobfuscated_entropy': deobfuscated_entropy,
            'new_strings_found': len(new_strings),
            'new_apis_found': len(new_apis),
            'deobfuscated_sha256': deobfuscated_sha256,
        }
        return self.deobfuscation_result

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type == "ClickHouseExporter":
            if not self.deobfuscation_result:
                return None

            r = self.deobfuscation_result
            current_time = datetime.now(timezone.utc)
            data = [[
                self.sha256,
                r['deobfuscator_used'],
                int(r['deobfuscation_successful']),
                r['original_size'],
                r['deobfuscated_size'],
                r['size_change_ratio'],
                r['original_entropy'],
                r['deobfuscated_entropy'],
                r['new_strings_found'],
                r['new_apis_found'],
                r['deobfuscated_sha256'],
                current_time,
            ]]

            column_names = [
                "sha256", "deobfuscator_used", "deobfuscation_successful",
                "original_size", "deobfuscated_size", "size_change_ratio",
                "original_entropy", "deobfuscated_entropy",
                "new_strings_found", "new_apis_found",
                "deobfuscated_sha256", "analysis_date",
            ]

            column_type_names = [
                "FixedString(64)", "LowCardinality(String)", "UInt8",
                "UInt64", "UInt64", "Float64",
                "Float64", "Float64",
                "UInt32", "UInt32",
                "FixedString(64)", "DateTime64(3, 'UTC')",
            ]

            return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "redb_js_deobfuscation"