Jan Christiansen

20 papers A 2C 6Misc 1Journal 5Unranked 5
YearRankTypeTitle / Venue / Authors
2023 C conf
PADL
Kai-Oliver Prott, Finn Teegen, Jan Christiansen
2020 J jnl
Theory Pract. Log. Program.
Sandra Dylus, Jan Christiansen, Finn Teegen
2019 J jnl
CoRR
Sandra Dylus, Jan Christiansen, Finn Teegen
2019 J jnl
Art Sci. Eng. Program.
Sandra Dylus, Jan Christiansen, Finn Teegen
2019 conf
Haskell@ICFP
Jan Christiansen, Sandra Dylus, Niels Bunkenburg
2018 J jnl
CoRR
Jan Christiansen, Sandra Dylus, Finn Teegen
2018 C conf
PADL
Sandra Dylus, Jan Christiansen, Finn Teegen
2016 A conf
ICFP
Jan Christiansen, Nikita Danilenko, Sandra Dylus
2013 C conf
PPDP
Jan Christiansen, Michael Hanus, Fabian Reck, Daniel Seidel
2011
Jan Christiansen
2011 C conf
PPDP
Jan Christiansen, Daniel Seidel
2011 C conf
PADL
Jan Christiansen
2010 conf
WFLP
Jan Christiansen, Daniel Seidel, Janis Voigtländer
2010 conf
PLPV
Jan Christiansen, Daniel Seidel, Janis Voigtländer
2009 J jnl
ACM SIGPLAN Notices
Jan Christiansen, Daniel Seidel, Janis Voigtländer
2008 conf
RelMiCS
Bernd Braßel, Jan Christiansen
2008 Misc conf
FLOPS
Jan Christiansen, Sebastian Fischer
2007 C conf
LOPSTR
Bernd Braßel, Jan Christiansen
2006 conf
Trends in Functional Programming
Jan Christiansen, Frank Huch
2004 A conf
ICFP
Jan Christiansen, Frank Huch
redb/extractors/js_extractors/js_content.py
← Index redb/extractors/js_extractors/js_content.py python
"""Persists raw + normalised text into the generic `code_text_content` table.

Reads the raw source and the deobfuscation result directly from the shared
JSContext so no extra compute happens here — both values are computed once
per sample (the source at JSContext construction, the deobfuscation lazily
on first access) and reused by any extractor that needs them.

`text_normalized` is left NULL when the deobfuscation pass produced no
output, so analysts can distinguish "we tried and got nothing" from
"normalisation succeeded".
"""

import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor


class JSContentExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.content_row = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_CONTENT.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, normalizer_used = self._context.deobfuscated

        self.content_row = {
            "content_type": self._context.content_type,
            "text_raw": src,
            "text_normalized": deobfuscated,  # may be None
            "normalizer_used": normalizer_used,  # may be None
        }
        return self.content_row

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type != "ClickHouseExporter":
            return None
        if not self.content_row:
            return None

        r = self.content_row
        data = [[
            self.sha256,
            r["content_type"],
            r["text_raw"],
            r["text_normalized"],
            r["normalizer_used"],
            datetime.now(timezone.utc),
        ]]

        column_names = [
            "sha256",
            "content_type",
            "text_raw",
            "text_normalized",
            "normalizer_used",
            "analysis_date",
        ]

        column_type_names = [
            "FixedString(64)",
            "LowCardinality(String)",
            "String",
            "Nullable(String)",
            "Nullable(String)",
            "DateTime64(3, 'UTC')",
        ]

        return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_text_content"