Jan Haltermann

20 papers A* 2A 1B 6C 5Journal 5
YearRankTypeTitle / Venue / Authors
2025
Jan Haltermann
2024 J jnl
Softw. Syst. Model.
Jan Haltermann, Heike Wehrheim
2024 J jnl
CoRR
Jan Haltermann, Marie-Christine Jakobs, Cedric Richter, Heike Wehrheim
2024 J jnl
Sci. Comput. Program.
Jan Haltermann, Marie-Christine Jakobs, Cedric Richter, Heike Wehrheim
2024 C conf
Software Engineering
Jan Haltermann, Marie-Christine Jakobs, Cedric Richter, Heike Wehrheim
2023 C conf
Software Engineering
Dirk Beyer, Jan Haltermann, Thomas Lemberger, Heike Wehrheim
2023 B conf
FASE
Jan Haltermann, Marie-Christine Jakobs, Cedric Richter, Heike Wehrheim
2023 B conf
SEFM
Jan Haltermann, Marie-Christine Jakobs, Cedric Richter, Heike Wehrheim
2023 B conf
SEFM
Nicola Thoben, Jan Haltermann, Heike Wehrheim
2023 C conf
Software Engineering
Cedric Richter, Jan Haltermann, Marie-Christine Jakobs, Felix Pauck, Stefan Schott, Heike Wehrheim
2022 A* conf
ASE
Cedric Richter, Jan Haltermann, Marie-Christine Jakobs, Felix Pauck, Stefan Schott, Heike Wehrheim
2022 C conf
Software Engineering
Jan Haltermann, Heike Wehrheim
2022 A* conf
ICSE
Dirk Beyer, Jan Haltermann, Thomas Lemberger, Heike Wehrheim
2022 B conf
SEFM
Jan Haltermann, Heike Wehrheim
2022 A conf
ICST
Jan Haltermann, Heike Wehrheim
2021 B conf
FASE
Jan Haltermann, Heike Wehrheim
2020 J jnl
CoRR
Jan Haltermann, Heike Wehrheim
2018 B conf
ARES
Kai Bemmann, Johannes Blömer, Jan Bobolz, Henrik Bröcher, Denis Diemert, Fabian Eidens, Lukas Eilers, Jan Haltermann, Jakob Juhnke, Burhan Otour, Laurens Porzenheim, Simon Pukrop, Erik Schilling, Michael Schlichtig, Marcel Stienemeier
2018 J jnl
IACR Cryptol. ePrint Arch.
Kai Bemmann, Johannes Blömer, Jan Bobolz, Henrik Bröcher, Denis Diemert, Fabian Eidens, Lukas Eilers, Jan Haltermann, Jakob Juhnke, Burhan Otour, Laurens Porzenheim, Simon Pukrop, Erik Schilling, Michael Schlichtig, Marcel Stienemeier
2018 C conf
ICTSS
Paul Börding, Jan Haltermann, Marie-Christine Jakobs, Heike Wehrheim
redb/extractors/js_extractors/js_content.py
← Index redb/extractors/js_extractors/js_content.py python
"""Persists raw + normalised text into the generic `code_text_content` table.

Reads the raw source and the deobfuscation result directly from the shared
JSContext so no extra compute happens here — both values are computed once
per sample (the source at JSContext construction, the deobfuscation lazily
on first access) and reused by any extractor that needs them.

`text_normalized` is left NULL when the deobfuscation pass produced no
output, so analysts can distinguish "we tried and got nothing" from
"normalisation succeeded".
"""

import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor


class JSContentExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.content_row = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_CONTENT.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, normalizer_used = self._context.deobfuscated

        self.content_row = {
            "content_type": self._context.content_type,
            "text_raw": src,
            "text_normalized": deobfuscated,  # may be None
            "normalizer_used": normalizer_used,  # may be None
        }
        return self.content_row

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type != "ClickHouseExporter":
            return None
        if not self.content_row:
            return None

        r = self.content_row
        data = [[
            self.sha256,
            r["content_type"],
            r["text_raw"],
            r["text_normalized"],
            r["normalizer_used"],
            datetime.now(timezone.utc),
        ]]

        column_names = [
            "sha256",
            "content_type",
            "text_raw",
            "text_normalized",
            "normalizer_used",
            "analysis_date",
        ]

        column_type_names = [
            "FixedString(64)",
            "LowCardinality(String)",
            "String",
            "Nullable(String)",
            "Nullable(String)",
            "DateTime64(3, 'UTC')",
        ]

        return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_text_content"