Ingo Karla

11 papers B 4Journal 2Unranked 5
YearRankTypeTitle / Venue / Authors
2013 conf
EW
Ingo Karla, János Bitó, Bernd Bochow, Ulrico Celentano, László Csurgai-Horváth, Pål Grønsund, Miguel López-Benítez, Ramiro Sámano-Robles
2011 conf
VTC Spring
Siegfried Klein, Ingo Karla, Edgar Kühn
2010 J jnl
Bell Labs Tech. J.
Christian G. Gerlach, Ingo Karla, Andreas Weber, Lutz Ewe, Hajo Bakker, Edgar Kuehn, Anil M. Rao
2009 J jnl
EURASIP J. Wirel. Commun. Netw.
Ingmar Blau, Gerhard Wunder, Ingo Karla, Rolf Sigle
2009 conf
MONAMI
Ingo Karla
2008 B conf
PIMRC
Ingmar Blau, Gerhard Wunder, Ingo Karla, Rolf Sigle
2007 B conf
PIMRC
Ingmar Blau, Gerhard Wunder, Ingo Karla, Rolf Sigle
2006 conf
VTC Spring
Mikael Prytz, Peter Karlsson, Catarina Cedervall, Aurelian Bria, Ingo Karla
2006 conf
VTC Fall
Joachim Sachs, Ramón Agüero, Miguel Berg, Jens Gebert, Ljupco Jorguseski, Ingo Karla, Peter Karlsson, George P. Koudouridis, Johan Lundsjö, Mikael Prytz, Ove Strandberg
2006 B conf
PIMRC
Guihua Piao, Klaus David, Ingo Karla, Rolf Sigle
2005 B conf
PIMRC
Fredrik Berggren, Aurelian Bria, Leonardo Badia, Ingo Karla, Remco Litjens, Per Magnusson, Francesco Meago, Haitao Tang, Riccardo Veronesi
redb/extractors/js_extractors/js_content.py
← Index redb/extractors/js_extractors/js_content.py python
"""Persists raw + normalised text into the generic `code_text_content` table.

Reads the raw source and the deobfuscation result directly from the shared
JSContext so no extra compute happens here — both values are computed once
per sample (the source at JSContext construction, the deobfuscation lazily
on first access) and reused by any extractor that needs them.

`text_normalized` is left NULL when the deobfuscation pass produced no
output, so analysts can distinguish "we tried and got nothing" from
"normalisation succeeded".
"""

import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor


class JSContentExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.content_row = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_CONTENT.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, normalizer_used = self._context.deobfuscated

        self.content_row = {
            "content_type": self._context.content_type,
            "text_raw": src,
            "text_normalized": deobfuscated,  # may be None
            "normalizer_used": normalizer_used,  # may be None
        }
        return self.content_row

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type != "ClickHouseExporter":
            return None
        if not self.content_row:
            return None

        r = self.content_row
        data = [[
            self.sha256,
            r["content_type"],
            r["text_raw"],
            r["text_normalized"],
            r["normalizer_used"],
            datetime.now(timezone.utc),
        ]]

        column_names = [
            "sha256",
            "content_type",
            "text_raw",
            "text_normalized",
            "normalizer_used",
            "analysis_date",
        ]

        column_type_names = [
            "FixedString(64)",
            "LowCardinality(String)",
            "String",
            "Nullable(String)",
            "Nullable(String)",
            "DateTime64(3, 'UTC')",
        ]

        return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_text_content"