Veli-Matti Karhulahti

16 papers C 4Journal 5Unranked 7
YearRankTypeTitle / Venue / Authors
2023 C conf
DiGRA
Kwok Ng, Raine Koskimaa, Veli-Matti Karhulahti, Miikka Sokka, Pauliina Husu, Sami Kokko, Pasi Koski
2021 J jnl
Games Cult.
Maria B. Garda, Veli-Matti Karhulahti
2020 J jnl
Int. J. Hum. Comput. Stud.
Jukka Vahlo, Veli-Matti Karhulahti
2018 conf
GamiFIN
Veli-Matti Karhulahti, Kai Kimppa
2017 C conf
FDG
Rune Kristian Lundedal Nielsen, Veli-Matti Karhulahti
2016 J jnl
Int. J. Gaming Comput. Mediat. Simulations
Tuomas Kari, Veli-Matti Karhulahti
2016 conf
DiGRA/FDG
Veli-Matti Karhulahti
2015 J jnl
Game Stud.
Veli-Matti Karhulahti
2014 conf
Nordic DiGRA
Jonne Arjoranta, Veli-Matti Karhulahti
2013 J jnl
Game Stud.
Veli-Matti Karhulahti
2013 C conf
FDG
Veli-Matti Karhulahti
2013 conf
DiGRA Conference
Veli-Matti Karhulahti
2012 conf
Fun and Games
Veli-Matti Karhulahti
2012 conf
Nordic DiGRA
Veli-Matti Karhulahti
2012 C conf
ICIDS
Veli-Matti Karhulahti
2011 conf
MindTrek
Veli-Matti Karhulahti
redb/extractors/js_extractors/js_content.py
← Index redb/extractors/js_extractors/js_content.py python
"""Persists raw + normalised text into the generic `code_text_content` table.

Reads the raw source and the deobfuscation result directly from the shared
JSContext so no extra compute happens here — both values are computed once
per sample (the source at JSContext construction, the deobfuscation lazily
on first access) and reused by any extractor that needs them.

`text_normalized` is left NULL when the deobfuscation pass produced no
output, so analysts can distinguish "we tried and got nothing" from
"normalisation succeeded".
"""

import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor


class JSContentExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.content_row = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_CONTENT.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, normalizer_used = self._context.deobfuscated

        self.content_row = {
            "content_type": self._context.content_type,
            "text_raw": src,
            "text_normalized": deobfuscated,  # may be None
            "normalizer_used": normalizer_used,  # may be None
        }
        return self.content_row

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type != "ClickHouseExporter":
            return None
        if not self.content_row:
            return None

        r = self.content_row
        data = [[
            self.sha256,
            r["content_type"],
            r["text_raw"],
            r["text_normalized"],
            r["normalizer_used"],
            datetime.now(timezone.utc),
        ]]

        column_names = [
            "sha256",
            "content_type",
            "text_raw",
            "text_normalized",
            "normalizer_used",
            "analysis_date",
        ]

        column_type_names = [
            "FixedString(64)",
            "LowCardinality(String)",
            "String",
            "Nullable(String)",
            "Nullable(String)",
            "DateTime64(3, 'UTC')",
        ]

        return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_text_content"