Maarten Hattink

13 papers C 1Journal 3Unranked 9
YearRankTypeTitle / Venue / Authors
2025 conf
OFC
Yuyang Wang, Songli Wang, Swarnava Sanyal, Nathaniel Nauman, Robert Parsons, James Robinson, Maarten Hattink, Kaylx Jang, Asher Novick, Karl J. McNulty, Xiang Meng, Michal Lipson, Alexander L. Gaeta, Keren Bergman
2024 conf
OFC
Asher Novick, Maarten Hattink, Anthony Rizzo, Yuyang Wang, Vignesh Gopal, Songli Wang, Robert Parsons, Keren Bergman
2023 conf
OFC
Vignesh Gopal, Anthony Rizzo, Maarten Hattink, Asher Novick, James Robinson, Kaveh Hosseini, Tim Tri Hoang, Keren Bergman
2022 J jnl
J. Parallel Distributed Comput.
Jorge González, Mauricio G. Palma, Maarten Hattink, Ruth Rubio-Noriega, Lois Orosa, Onur Mutlu, Keren Bergman, Rodolfo Azevedo
2022 conf
OFC
Maarten Hattink, Liang Yuan Dai, Ziyi Zhu, Keren Bergman
2021 J jnl
IEEE Internet Comput.
Fred Douglis, Seth Robertson, Eric van den Berg, Josephine Micallef, Marc Pucci, Alex Aiken, Keren Bergman, Maarten Hattink, Mingoo Seok
2020 conf
HPEC
Maarten Hattink, Giuseppe Di Guglielmo, Luca P. Carloni, Keren Bergman
2020 C conf
SBAC-PAD
Jorge González, Alexander Gazman, Maarten Hattink, Mauricio G. Palma, Meisam Bahadori, Ruth Rubio-Noriega, Lois Orosa, Madeleine Glick, Onur Mutlu, Keren Bergman, Rodolfo Azevedo
2020 J jnl
CoRR
Jorge González, Alexander Gazman, Maarten Hattink, Mauricio G. Palma, Meisam Bahadori, Ruth Rubio-Noriega, Lois Orosa, Madeleine Glick, Onur Mutlu, Keren Bergman, Rodolfo Azevedo
2019 conf
OFC
Ziyi Zhu, Yiwen Shen, Yishen Huang, Alexander Gazman, Maarten Hattink, Keren Bergman
2018 conf
OFC
Yiwen Shen, Alexander Gazman, Ziyi Zhu, Min Yee Teh, Maarten Hattink, Sébastien Rumley, Payman Samadi, Keren Bergman
2018 conf
OFC
Erik F. Anderson, Alexander Gazman, Ziyi Zhu, Maarten Hattink, Keren Bergman
2017 conf
ECOC
Yiwen Shen, Payman Samadi, Ziyi Zhu, Alexander Gazman, Erik F. Anderson, David M. Calhoun, Maarten Hattink, Keren Bergman
redb/extractors/js_extractors/js_content.py
← Index redb/extractors/js_extractors/js_content.py python
"""Persists raw + normalised text into the generic `code_text_content` table.

Reads the raw source and the deobfuscation result directly from the shared
JSContext so no extra compute happens here — both values are computed once
per sample (the source at JSContext construction, the deobfuscation lazily
on first access) and reused by any extractor that needs them.

`text_normalized` is left NULL when the deobfuscation pass produced no
output, so analysts can distinguish "we tried and got nothing" from
"normalisation succeeded".
"""

import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor


class JSContentExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.content_row = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_CONTENT.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, normalizer_used = self._context.deobfuscated

        self.content_row = {
            "content_type": self._context.content_type,
            "text_raw": src,
            "text_normalized": deobfuscated,  # may be None
            "normalizer_used": normalizer_used,  # may be None
        }
        return self.content_row

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type != "ClickHouseExporter":
            return None
        if not self.content_row:
            return None

        r = self.content_row
        data = [[
            self.sha256,
            r["content_type"],
            r["text_raw"],
            r["text_normalized"],
            r["normalizer_used"],
            datetime.now(timezone.utc),
        ]]

        column_names = [
            "sha256",
            "content_type",
            "text_raw",
            "text_normalized",
            "normalizer_used",
            "analysis_date",
        ]

        column_type_names = [
            "FixedString(64)",
            "LowCardinality(String)",
            "String",
            "Nullable(String)",
            "Nullable(String)",
            "DateTime64(3, 'UTC')",
        ]

        return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_text_content"