James Cooper

12 papers A 1C 1Misc 3Journal 5Unranked 2
YearRankTypeTitle / Venue / Authors
2024 J jnl
CoRR
Zeheng Wang, James Cooper, Muhammad Usman, Timothy van der Laan
2022 J jnl
Axioms
Radu Nicolescu, Michael J. Dinneen, James Cooper, Alec Henderson, Yezhou Liu
2022 J jnl
J. Membr. Comput.
James Cooper, Radu Nicolescu
2022 C conf
IDEAL
James Cooper, Peter Mitic, Gesine Reinert, Tadas Temcinas
2020 conf
IDEAL (2)
Peter Mitic, James Cooper
2019 J jnl
J. Membr. Comput.
James Cooper, Radu Nicolescu
2019 J jnl
Fundam. Informaticae
James Cooper, Radu Nicolescu
2018 Misc conf
IVCNZ
James Cooper
2017 Misc conf
MVA
James Cooper, Mihailo Azhar, Trevor Gee, Wannes van der Mark, Patrice Delmas, Georgy L. Gimel'farb
2017 Misc conf
IVCNZ
A. Anderson, Mihailo Azhar, James Cooper, Jason James, J. Debes, D. Azhar, W. Vandermark, K.-C. Leung, K. Yang, J. Hilman, Stefano Schenone, Alfonso Gastelum Strozzi, Trevor Gee, Heide Friedrich, Simon F. Thrush, Patrice Delmas
2017 A conf
AIED
Mark A. Riedesel, Neil L. Zimmerman, Ryan S. Baker, Tom Titchener, James Cooper
2004 conf
ISBI
Joy Dunkers, Forrest Landis, Marcus Cicerone, James Cooper, Newell Washburn
redb/extractors/js_extractors/js_content.py
← Index redb/extractors/js_extractors/js_content.py python
"""Persists raw + normalised text into the generic `code_text_content` table.

Reads the raw source and the deobfuscation result directly from the shared
JSContext so no extra compute happens here — both values are computed once
per sample (the source at JSContext construction, the deobfuscation lazily
on first access) and reused by any extractor that needs them.

`text_normalized` is left NULL when the deobfuscation pass produced no
output, so analysts can distinguish "we tried and got nothing" from
"normalisation succeeded".
"""

import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor


class JSContentExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.content_row = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_CONTENT.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, normalizer_used = self._context.deobfuscated

        self.content_row = {
            "content_type": self._context.content_type,
            "text_raw": src,
            "text_normalized": deobfuscated,  # may be None
            "normalizer_used": normalizer_used,  # may be None
        }
        return self.content_row

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type != "ClickHouseExporter":
            return None
        if not self.content_row:
            return None

        r = self.content_row
        data = [[
            self.sha256,
            r["content_type"],
            r["text_raw"],
            r["text_normalized"],
            r["normalizer_used"],
            datetime.now(timezone.utc),
        ]]

        column_names = [
            "sha256",
            "content_type",
            "text_raw",
            "text_normalized",
            "normalizer_used",
            "analysis_date",
        ]

        column_type_names = [
            "FixedString(64)",
            "LowCardinality(String)",
            "String",
            "Nullable(String)",
            "Nullable(String)",
            "DateTime64(3, 'UTC')",
        ]

        return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_text_content"