Caroline Colijn

18 papers Journal 18
YearRankTypeTitle / Venue / Authors
2024 J jnl
CoRR
Cédric Chauve, Caroline Colijn, Louxin Zhang
2024 J jnl
PLoS Comput. Biol.
Niloufar Abhari, Caroline Colijn, Arne Mooers, Paul F. Tupper
2023 J jnl
J. Comput. Biol.
Shijia Wang, Shufei Ge, Benjamin Sobkowiak, Liangliang Wang, Louis Grandjean, Caroline Colijn, Lloyd T. Elliott
2023 J jnl
PLoS Comput. Biol.
Nicola Mulberry, Alexander R. Rutherford, Caroline Colijn
2021 J jnl
PLoS Comput. Biol.
Paul F. Tupper, Caroline Colijn
2021 J jnl
J. Comput. Biol.
Shijia Wang, Shufei Ge, Caroline Colijn, Priscila Biller, Liangliang Wang, Lloyd T. Elliott
2020 J jnl
Entropy
Cornelia Metzig, Caroline Colijn
2020 J jnl
Pattern Recognit. Lett.
Cornelia Metzig, Matthew Gould, Roshan Noronha, Roshani Abbey, Mark Sandler, Caroline Colijn
2020 J jnl
PLoS Comput. Biol.
Sean C. Anderson, Andrew M. Edwards, Madi Yerlanov, Nicola Mulberry, Jessica E. Stockdale, Sarafa A. Iyaniwura, Rebeca C. Falcao, Michael C. Otterstatter, Michael A. Irvine, Naveed Z. Janjua, Daniel Coombs, Caroline Colijn
2019 J jnl
PLoS Comput. Biol.
Cornelia Metzig, Oliver Ratmann, Daniela Bezemer, Caroline Colijn
2017 J jnl
PLoS Comput. Biol.
Nick Fyson, Jerry King, Thomas Belcher, Andrew Preston, Caroline Colijn
2017 J jnl
PLoS Comput. Biol.
Don Klinkenberg, Jantien A. Backer, Xavier Didelot, Caroline Colijn, Jacco Wallinga
2016 J jnl
J. Appl. Probab.
Giacomo Plazzotta, Caroline Colijn
2016 J jnl
PLoS Comput. Biol.
Leonid Chindelevitch, Caroline Colijn, Prashini Moodley, Douglas Wilson, Ted Cohen
2013 J jnl
PLoS Comput. Biol.
Katy Robinson, Nick Fyson, Ted Cohen, Christophe Fraser, Caroline Colijn
2011 J jnl
BMC Syst. Biol.
Kuhn Ip, Caroline Colijn, Desmond S. Lun
2009 J jnl
PLoS Comput. Biol.
Caroline Colijn, Aaron Brandes, Jeremy Zucker, Desmond S. Lun, Brian Weiner, Maha R. Farhat, Tan-Yun Cheng, D. Branch Moody, Megan Murray, James E. Galagan
2007 J jnl
SIAM J. Appl. Dyn. Syst.
Caroline Colijn, Michael C. Mackey
redb/extractors/js_extractors/js_content.py
← Index redb/extractors/js_extractors/js_content.py python
"""Persists raw + normalised text into the generic `code_text_content` table.

Reads the raw source and the deobfuscation result directly from the shared
JSContext so no extra compute happens here — both values are computed once
per sample (the source at JSContext construction, the deobfuscation lazily
on first access) and reused by any extractor that needs them.

`text_normalized` is left NULL when the deobfuscation pass produced no
output, so analysts can distinguish "we tried and got nothing" from
"normalisation succeeded".
"""

import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor


class JSContentExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.content_row = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_CONTENT.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, normalizer_used = self._context.deobfuscated

        self.content_row = {
            "content_type": self._context.content_type,
            "text_raw": src,
            "text_normalized": deobfuscated,  # may be None
            "normalizer_used": normalizer_used,  # may be None
        }
        return self.content_row

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type != "ClickHouseExporter":
            return None
        if not self.content_row:
            return None

        r = self.content_row
        data = [[
            self.sha256,
            r["content_type"],
            r["text_raw"],
            r["text_normalized"],
            r["normalizer_used"],
            datetime.now(timezone.utc),
        ]]

        column_names = [
            "sha256",
            "content_type",
            "text_raw",
            "text_normalized",
            "normalizer_used",
            "analysis_date",
        ]

        column_type_names = [
            "FixedString(64)",
            "LowCardinality(String)",
            "String",
            "Nullable(String)",
            "Nullable(String)",
            "DateTime64(3, 'UTC')",
        ]

        return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_text_content"