Raffaele Scapellato

18 papers Journal 18
YearRankTypeTitle / Venue / Authors
2019 J jnl
Discret. Appl. Math.
Josef Lauri, Russell Mizzi, Raffaele Scapellato
2015 J jnl
Ars Math. Contemp.
Josef Lauri, Russell Mizzi, Raffaele Scapellato
2011 J jnl
Australas. J Comb.
Josef Lauri, Russell Mizzi, Raffaele Scapellato
2009 J jnl
Discret. Appl. Math.
Raffaele Scapellato, Ramadan A. El-Shanawany, Mahmoud Shibl Higazy
2009 J jnl
Australas. J Comb.
Ramadan A. El-Shanawany, Mahmoud Shibl Higazy, Raffaele Scapellato
2002 J jnl
Discret. Math.
Paolo Dulio, Raffaele Scapellato
1998 J jnl
Eur. J. Comb.
Dragan Marusic, Raffaele Scapellato
1997 J jnl
Discret. Math.
Walter Pacco, Raffaele Scapellato
1994 J jnl
Discret. Math.
Dragan Marusic, Raffaele Scapellato
1994 J jnl
Comb.
Dragan Marusic, Raffaele Scapellato
1994 J jnl
Discret. Math.
Dragan Marusic, Raffaele Scapellato
1993 J jnl
Discret. Math.
Raffaele Scapellato
1993 J jnl
J. Comb. Theory B
Dragan Marusic, Raffaele Scapellato
1992 J jnl
Discret. Math.
Dragan Marusic, Raffaele Scapellato
1992 J jnl
J. Graph Theory
Dragan Marusic, Raffaele Scapellato
1992 J jnl
Discret. Math.
Dragan Marusic, Raffaele Scapellato, Norma Zagaglia Salvi
1990 J jnl
Discret. Math.
Raffaele Scapellato
1986 J jnl
J. Comb. Theory B
Raffaele Scapellato
redb/extractors/js_extractors/js_content.py
← Index redb/extractors/js_extractors/js_content.py python
"""Persists raw + normalised text into the generic `code_text_content` table.

Reads the raw source and the deobfuscation result directly from the shared
JSContext so no extra compute happens here — both values are computed once
per sample (the source at JSContext construction, the deobfuscation lazily
on first access) and reused by any extractor that needs them.

`text_normalized` is left NULL when the deobfuscation pass produced no
output, so analysts can distinguish "we tried and got nothing" from
"normalisation succeeded".
"""

import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor


class JSContentExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.content_row = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_CONTENT.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, normalizer_used = self._context.deobfuscated

        self.content_row = {
            "content_type": self._context.content_type,
            "text_raw": src,
            "text_normalized": deobfuscated,  # may be None
            "normalizer_used": normalizer_used,  # may be None
        }
        return self.content_row

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type != "ClickHouseExporter":
            return None
        if not self.content_row:
            return None

        r = self.content_row
        data = [[
            self.sha256,
            r["content_type"],
            r["text_raw"],
            r["text_normalized"],
            r["normalizer_used"],
            datetime.now(timezone.utc),
        ]]

        column_names = [
            "sha256",
            "content_type",
            "text_raw",
            "text_normalized",
            "normalizer_used",
            "analysis_date",
        ]

        column_type_names = [
            "FixedString(64)",
            "LowCardinality(String)",
            "String",
            "Nullable(String)",
            "Nullable(String)",
            "DateTime64(3, 'UTC')",
        ]

        return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_text_content"