Ian Craw

17 papers A* 1A 10B 1Journal 4Unranked 1
YearRankTypeTitle / Venue / Authors
2001 A conf
BMVC
David A. Brown, Ian Craw, Julian Lewthwaite
1999 J jnl
IEEE Trans. Pattern Anal. Mach. Intell.
Ian Craw, Nicholas Costen, Takashi Kato, Shigeru Akamatsu
1996 conf
ECCV (1)
Nicholas Costen, Ian Craw, Graham Robertson, Shigeru Akamatsu
1996 B conf
FG
Nicholas Costen, Takashi Kato, Shigeru Akamatsu, Ian Craw, Graham Robertson
1996 J jnl
Image Vis. Comput.
David Tock, Ian Craw
1995 A conf
BMVC
David Tock, Ian Craw
1994 A conf
BMVC
Graham Robertson, Ian Craw, Blair Donaldson
1994 J jnl
Image Vis. Comput.
Graham Robertson, Ian Craw
1993 A conf
BMVC
Graham Robertson, Ian Craw
1992 A conf
BMVC
David Tock, Ian Craw
1992 A conf
BMVC
Ian Craw, Peter Cameron
1992 A* conf
ECCV
Ian Craw, David Tock, Alan Bennett
1991 A conf
BMVC
Alan Bennett, Ian Craw
1991 A conf
BMVC
Ian Craw, Peter Cameron
1991 A conf
BMVC
Andrew C. Aitchison, Ian Craw
1990 A conf
BMVC
David Tock, Ian Craw, Roly Lishman
1987 J jnl
Pattern Recognit. Lett.
Ian Craw, H. Ellis, J. Rowland Lishman
redb/extractors/js_extractors/js_content.py
← Index redb/extractors/js_extractors/js_content.py python
"""Persists raw + normalised text into the generic `code_text_content` table.

Reads the raw source and the deobfuscation result directly from the shared
JSContext so no extra compute happens here — both values are computed once
per sample (the source at JSContext construction, the deobfuscation lazily
on first access) and reused by any extractor that needs them.

`text_normalized` is left NULL when the deobfuscation pass produced no
output, so analysts can distinguish "we tried and got nothing" from
"normalisation succeeded".
"""

import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor


class JSContentExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.content_row = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_CONTENT.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, normalizer_used = self._context.deobfuscated

        self.content_row = {
            "content_type": self._context.content_type,
            "text_raw": src,
            "text_normalized": deobfuscated,  # may be None
            "normalizer_used": normalizer_used,  # may be None
        }
        return self.content_row

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type != "ClickHouseExporter":
            return None
        if not self.content_row:
            return None

        r = self.content_row
        data = [[
            self.sha256,
            r["content_type"],
            r["text_raw"],
            r["text_normalized"],
            r["normalizer_used"],
            datetime.now(timezone.utc),
        ]]

        column_names = [
            "sha256",
            "content_type",
            "text_raw",
            "text_normalized",
            "normalizer_used",
            "analysis_date",
        ]

        column_type_names = [
            "FixedString(64)",
            "LowCardinality(String)",
            "String",
            "Nullable(String)",
            "Nullable(String)",
            "DateTime64(3, 'UTC')",
        ]

        return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_text_content"