Catherine Hall

15 papers A 1B 1Journal 3Unranked 10
YearRankTypeTitle / Venue / Authors
2016 J jnl
Coll. Res. Libr.
Michael J. Khoo, Lily Rozaklis, Catherine Hall, Diana S. Kusunoki
2013 conf
ASIST
Michael Khoo, Lily Rozaklis, Catherine Hall, Diana S. Kusunoki
2013 J jnl
Inf. Organ.
Michael Khoo, Catherine Hall
2013 conf
JCDL
Catherine Hall, Michael Khoo
2013 A conf
ICWSM
Michael A. Zarro, Catherine Hall, Andrea Forte
2012 J jnl
D Lib Mag.
Michael A. Zarro, Catherine Hall
2012 conf
JCDL
Michael A. Zarro, Catherine Hall
2012 conf
ASIST
Catherine Hall, Michael A. Zarro
2012 B conf
TPDL
Michael Khoo, Catherine Hall
2011 conf
ASIST
Catherine Hall
2011 conf
CTS
Catherine Hall, Mi Zhang
2011 conf
JCDL
Catherine Hall, Michael A. Zarro
2010 conf
ICADL
Robert B. Allen, Catherine Hall
2010 conf
JCDL
Michael Khoo, Catherine Hall
2009 conf
ASIST
Catherine Hall, Robin Naughton, Xia Lin
redb/extractors/js_extractors/js_content.py
← Index redb/extractors/js_extractors/js_content.py python
"""Persists raw + normalised text into the generic `code_text_content` table.

Reads the raw source and the deobfuscation result directly from the shared
JSContext so no extra compute happens here — both values are computed once
per sample (the source at JSContext construction, the deobfuscation lazily
on first access) and reused by any extractor that needs them.

`text_normalized` is left NULL when the deobfuscation pass produced no
output, so analysts can distinguish "we tried and got nothing" from
"normalisation succeeded".
"""

import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor


class JSContentExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.content_row = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_CONTENT.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, normalizer_used = self._context.deobfuscated

        self.content_row = {
            "content_type": self._context.content_type,
            "text_raw": src,
            "text_normalized": deobfuscated,  # may be None
            "normalizer_used": normalizer_used,  # may be None
        }
        return self.content_row

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type != "ClickHouseExporter":
            return None
        if not self.content_row:
            return None

        r = self.content_row
        data = [[
            self.sha256,
            r["content_type"],
            r["text_raw"],
            r["text_normalized"],
            r["normalizer_used"],
            datetime.now(timezone.utc),
        ]]

        column_names = [
            "sha256",
            "content_type",
            "text_raw",
            "text_normalized",
            "normalizer_used",
            "analysis_date",
        ]

        column_type_names = [
            "FixedString(64)",
            "LowCardinality(String)",
            "String",
            "Nullable(String)",
            "Nullable(String)",
            "DateTime64(3, 'UTC')",
        ]

        return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_text_content"