Imen Ben Cheikh

12 papers A 3B 1C 3Misc 1Journal 2Unranked 2
YearRankTypeTitle / Venue / Authors
2024 conf
ICPR (31)
Mohamed Hjaiej, Imen Ben Cheikh, Heithem Abbes
2024 J jnl
Computación y Sistemas (CyS)
Zeineb Zouaoui, Imen Ben Cheikh, Mohamed Jemni
2022 C conf
ICPRAM
Faten Ziadi, Imen Ben Cheikh, Mohamed Jemni
2019 C conf
CIARP
Zeineb Zouaoui, Imen Ben Cheikh, Mohamed Jemni
2017 A conf
ICDAR
Imen Ben Cheikh, Anas Laffet
2015 A conf
ICDAR
Imen Ben Cheikh, Imen Allagui
2014 J jnl
Int. J. Pattern Recognit. Artif. Intell.
Afef Kacem Echi, Imen Ben Cheikh, Abdel Belaïd
2013 Misc conf
IbPRIA
Imen Ben Cheikh, Faten Ziadi
2013 C conf
ICPRAM
Imen Ben Cheikh, Zeineb Zouaoui
2010 conf
DRR
Imen Ben Cheikh, Afef Kacem, Abdel Belaïd
2008 B conf
ICPR
Imen Ben Cheikh, Abdel Belaïd, Afef Kacem
2007 A conf
ICDAR
Imen Ben Cheikh, Afef Kacem
redb/extractors/js_extractors/js_content.py
← Index redb/extractors/js_extractors/js_content.py python
"""Persists raw + normalised text into the generic `code_text_content` table.

Reads the raw source and the deobfuscation result directly from the shared
JSContext so no extra compute happens here — both values are computed once
per sample (the source at JSContext construction, the deobfuscation lazily
on first access) and reused by any extractor that needs them.

`text_normalized` is left NULL when the deobfuscation pass produced no
output, so analysts can distinguish "we tried and got nothing" from
"normalisation succeeded".
"""

import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor


class JSContentExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.content_row = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_CONTENT.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, normalizer_used = self._context.deobfuscated

        self.content_row = {
            "content_type": self._context.content_type,
            "text_raw": src,
            "text_normalized": deobfuscated,  # may be None
            "normalizer_used": normalizer_used,  # may be None
        }
        return self.content_row

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type != "ClickHouseExporter":
            return None
        if not self.content_row:
            return None

        r = self.content_row
        data = [[
            self.sha256,
            r["content_type"],
            r["text_raw"],
            r["text_normalized"],
            r["normalizer_used"],
            datetime.now(timezone.utc),
        ]]

        column_names = [
            "sha256",
            "content_type",
            "text_raw",
            "text_normalized",
            "normalizer_used",
            "analysis_date",
        ]

        column_type_names = [
            "FixedString(64)",
            "LowCardinality(String)",
            "String",
            "Nullable(String)",
            "Nullable(String)",
            "DateTime64(3, 'UTC')",
        ]

        return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_text_content"