Jacob Stein

22 papers A* 2A 4B 1Journal 4Unranked 7
YearRankTypeTitle / Venue / Authors
2018 J jnl
Appl. Clin. Inform.
Jacob Stein, Jared Klein, Thomas H. Payne, Sara Jackson, Sue Peacock, Natalia Oster, Trinell Carpenter, Joann G. Elmore
2014 B conf
SPLC
Jacob Stein, Ingrid Nunes, Elder Cirilo
1994 A conf
OOPSLA
Thomas Atwoode, Jnan Dash, Jacob Stein, Michael Stonebraker, Mary E. S. Loomis
1991 ch.
On Object-Oriented Database System
Jacob Stein, David Maier
1991 J jnl
Commun. ACM
Paul Butterworth, Allen Otis, Jacob Stein
1990 ch.
Research Foundations in Object-Oriented and Semantic Database Systems
David Maier, Jacob Stein, Allen Otis, Alan Purdy
1990 conf
OOPSLA/ECOOP
Jacob Stein, Tim Andrews, Bill Kent, Kate Rotzell, Daniel Weinreb
1990 conf
OOPSLA/ECOOP Addendum
Jacob Stein, Tim Andrews, Bill Kent, Mary Lumas, Daniel Weinreb
1989 conf
DBPL
Jacob Stein, T. Lougenia Anderson, David Maier
1989 A conf
OOPSLA
Alan Purdy, Jeff Sutherland, Mike Caruso, Tom Atwood, Tim Andrews, Jacob Stein
1989 ch.
Object-Oriented Concepts, Databases, and Applications
Robert Bretl, David Maier, Allen Otis, D. Jason Penney, Bruce Schuchardt, Jacob Stein, E. Harold Williams, Monty Williams
1987 A conf
OOPSLA
D. Jason Penney, Jacob Stein
1987 ch.
Research Directions in Object-Oriented Programming
David Maier, Jacob Stein
1987 conf
POS
D. Jason Penney, Jacob Stein, David Maier
1987 J jnl
Inf. Syst.
David Maier, David Rozenshtein, Sharon C. Salveter, Jacob Stein, David Scott Warren
1986 A conf
OOPSLA
David Maier, Jacob Stein, Allen Otis, Alan Purdy
1986 conf
XP7.52 Workshop on Database Theory
David Maier, Jacob Stein, Allen Otis, Alan Purdy
1986 conf
OODBS
David Maier, Jacob Stein
1985 A* conf
PODS
Jacob Stein, David Maier
1985 J jnl
IEEE Trans. Software Eng.
David Maier, David Rozenshtein, Jacob Stein
1984 A* conf
ICDE
David Maier, David Rozenshtein, Jacob Stein
1982 conf
SIGMOD Conference
David Maier, David Rozenshtein, Sharon C. Salveter, Jacob Stein, David Scott Warren
redb/extractors/js_extractors/js_content.py
← Index redb/extractors/js_extractors/js_content.py python
"""Persists raw + normalised text into the generic `code_text_content` table.

Reads the raw source and the deobfuscation result directly from the shared
JSContext so no extra compute happens here — both values are computed once
per sample (the source at JSContext construction, the deobfuscation lazily
on first access) and reused by any extractor that needs them.

`text_normalized` is left NULL when the deobfuscation pass produced no
output, so analysts can distinguish "we tried and got nothing" from
"normalisation succeeded".
"""

import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor


class JSContentExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.content_row = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_CONTENT.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, normalizer_used = self._context.deobfuscated

        self.content_row = {
            "content_type": self._context.content_type,
            "text_raw": src,
            "text_normalized": deobfuscated,  # may be None
            "normalizer_used": normalizer_used,  # may be None
        }
        return self.content_row

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type != "ClickHouseExporter":
            return None
        if not self.content_row:
            return None

        r = self.content_row
        data = [[
            self.sha256,
            r["content_type"],
            r["text_raw"],
            r["text_normalized"],
            r["normalizer_used"],
            datetime.now(timezone.utc),
        ]]

        column_names = [
            "sha256",
            "content_type",
            "text_raw",
            "text_normalized",
            "normalizer_used",
            "analysis_date",
        ]

        column_type_names = [
            "FixedString(64)",
            "LowCardinality(String)",
            "String",
            "Nullable(String)",
            "Nullable(String)",
            "DateTime64(3, 'UTC')",
        ]

        return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_text_content"