Ozgur Akgun

13 papers A* 1A 2B 1C 2Journal 6Unranked 1
YearRankTypeTitle / Venue / Authors
2025 J jnl
CoRR
Ozgur Akgun, Mun See Chang, Ian P. Gent, Christopher Jefferson
2019 J jnl
IEEE Trans. Cloud Comput.
Blesson Varghese, Ozgur Akgun, Ian Miguel, Long Thai, Adam Barker
2018 conf
ICDM Workshops
Gokberk Kocak, Ozgur Akgun, Ian Miguel, Peter Nightingale
2016 J jnl
CoRR
Blesson Varghese, Ozgur Akgun, Ian Miguel, Long Thai, Adam Barker
2015 B conf
e-Science
James Wetter, Ozgur Akgun, Adam Barker, Martin Dominik, Ian Miguel, Blesson Varghese
2014 A conf
ECAI
Ozgur Akgun, Ian P. Gent, Christopher Jefferson, Ian Miguel, Peter Nightingale
2014 C conf
CloudCom
Blesson Varghese, Ozgur Akgun, Ian Miguel, Long Thai, Adam Barker
2014 J jnl
CoRR
Blesson Varghese, Ozgur Akgun, Ian Miguel, Long Thai, Adam Barker
2014 C conf
CloudCom
Long Thai, Adam Barker, Blesson Varghese, Ozgur Akgun, Ian Miguel
2014 J jnl
CoRR
Long Thai, Adam Barker, Blesson Varghese, Ozgur Akgun, Ian Miguel
2013 A conf
CP
Ozgur Akgun, Alan M. Frisch, Ian P. Gent, Bilal Syed Hussain, Christopher Jefferson, Lars Kotthoff, Ian Miguel, Peter Nightingale
2011 J jnl
CoRR
Ozgur Akgun, Alan M. Frisch, Brahim Hnich, Christopher Jefferson, Ian Miguel
2011 A* conf
AAAI
Ozgur Akgun, Ian Miguel, Christopher Jefferson, Alan M. Frisch, Brahim Hnich
redb/extractors/js_extractors/js_content.py
← Index redb/extractors/js_extractors/js_content.py python
"""Persists raw + normalised text into the generic `code_text_content` table.

Reads the raw source and the deobfuscation result directly from the shared
JSContext so no extra compute happens here — both values are computed once
per sample (the source at JSContext construction, the deobfuscation lazily
on first access) and reused by any extractor that needs them.

`text_normalized` is left NULL when the deobfuscation pass produced no
output, so analysts can distinguish "we tried and got nothing" from
"normalisation succeeded".
"""

import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor


class JSContentExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.content_row = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_CONTENT.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, normalizer_used = self._context.deobfuscated

        self.content_row = {
            "content_type": self._context.content_type,
            "text_raw": src,
            "text_normalized": deobfuscated,  # may be None
            "normalizer_used": normalizer_used,  # may be None
        }
        return self.content_row

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type != "ClickHouseExporter":
            return None
        if not self.content_row:
            return None

        r = self.content_row
        data = [[
            self.sha256,
            r["content_type"],
            r["text_raw"],
            r["text_normalized"],
            r["normalizer_used"],
            datetime.now(timezone.utc),
        ]]

        column_names = [
            "sha256",
            "content_type",
            "text_raw",
            "text_normalized",
            "normalizer_used",
            "analysis_date",
        ]

        column_type_names = [
            "FixedString(64)",
            "LowCardinality(String)",
            "String",
            "Nullable(String)",
            "Nullable(String)",
            "DateTime64(3, 'UTC')",
        ]

        return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_text_content"