James L. Szalma

18 papers Journal 16Unranked 1
YearRankTypeTitle / Venue / Authors
2021 J jnl
Hum. Factors
Peter A. Hancock, Theresa T. Kessler, Alexandra D. Kaplan, John C. Brill, James L. Szalma
2019 J jnl
Comput. Hum. Behav.
Victoria L. Claypoole, James L. Szalma
2019 J jnl
Hum. Factors
Peter A. Hancock, James L. Szalma
2019 J jnl
Hum. Factors
Alexis R. Neigel, Daryn A. Dever, Victoria L. Claypoole, James L. Szalma
2019 J jnl
Hum. Factors
Victoria L. Claypoole, Daryn A. Dever, Kody L. Denues, James L. Szalma
2019 J jnl
Hum. Factors
Ryan W. Wohleber, Gerald Matthews, Jinchao Lin, James L. Szalma, Gloria L. Calhoun, Gregory J. Funke, C.-Y. Peter Chiu, Heath A. Ruff
2018 J jnl
IEEE Trans. Hum. Mach. Syst.
Jennifer E. Thropp, Tal Oron-Gilad, James L. Szalma, Peter A. Hancock
2018 J jnl
Hum. Factors
Victoria L. Claypoole, James L. Szalma
2017 J jnl
Hum. Factors
Peter A. Hancock, Carryl L. Baldwin, Joel S. Warm, James L. Szalma
2017 conf
HCI (15)
Tarah Daly, Jennifer Murphy, Katlin Anglin, James L. Szalma, Max Acree, Carla Landsberg, Laticia Bowens
2016 J jnl
Hum. Factors
Kristin E. Schaefer, Jessie Y. C. Chen, James L. Szalma, Peter A. Hancock
2016 J jnl
Comput. Hum. Behav.
Shan G. Lakhmani, Paul Oppold, Michael A. Rupp, James L. Szalma, Peter A. Hancock
2014 J jnl
Hum. Factors
James L. Szalma
2007 J jnl
Hum. Factors
Peter A. Hancock, Jennifer M. Ross, James L. Szalma
2006 ch.
Neuroergonomics
Peter A. Hancock, James L. Szalma
2006 J jnl
Hum. Factors
James L. Szalma, Peter A. Hancock, Joel S. Warm, William N. Dember, Kelley S. Parsons
2004 J jnl
Hum. Factors
James L. Szalma, Joel S. Warm, Gerald Matthews, William N. Dember, Ernest M. Weiler, Ashley Meier, F. Thomas Eggemeier
2003 J jnl
Hum. Factors
Rebecca A. Grier, Joel S. Warm, William N. Dember, Gerald Matthews, Traci L. Galinsky, James L. Szalma, Raja Parasuraman
redb/extractors/js_extractors/js_content.py
← Index redb/extractors/js_extractors/js_content.py python
"""Persists raw + normalised text into the generic `code_text_content` table.

Reads the raw source and the deobfuscation result directly from the shared
JSContext so no extra compute happens here — both values are computed once
per sample (the source at JSContext construction, the deobfuscation lazily
on first access) and reused by any extractor that needs them.

`text_normalized` is left NULL when the deobfuscation pass produced no
output, so analysts can distinguish "we tried and got nothing" from
"normalisation succeeded".
"""

import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor


class JSContentExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.content_row = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_CONTENT.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, normalizer_used = self._context.deobfuscated

        self.content_row = {
            "content_type": self._context.content_type,
            "text_raw": src,
            "text_normalized": deobfuscated,  # may be None
            "normalizer_used": normalizer_used,  # may be None
        }
        return self.content_row

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type != "ClickHouseExporter":
            return None
        if not self.content_row:
            return None

        r = self.content_row
        data = [[
            self.sha256,
            r["content_type"],
            r["text_raw"],
            r["text_normalized"],
            r["normalizer_used"],
            datetime.now(timezone.utc),
        ]]

        column_names = [
            "sha256",
            "content_type",
            "text_raw",
            "text_normalized",
            "normalizer_used",
            "analysis_date",
        ]

        column_type_names = [
            "FixedString(64)",
            "LowCardinality(String)",
            "String",
            "Nullable(String)",
            "Nullable(String)",
            "DateTime64(3, 'UTC')",
        ]

        return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_text_content"