Olaf Chitil

37 papers A* 2A 3B 1C 5Misc 1Journal 4Unranked 16
YearRankTypeTitle / Venue / Authors
2021 conf
IFL
Joanna Sharrad, Olaf Chitil
2020 ed.
IFL
Olaf Chitil
2020 conf
TFP
Joanna Sharrad, Olaf Chitil
2019 conf
IFL
Kanae Tsushima, Olaf Chitil, Joanna Sharrad
2018 Misc conf
FLOPS
Kanae Tsushima, Olaf Chitil
2018 conf
IFL
Joanna Sharrad, Olaf Chitil, Meng Wang
2018 J jnl
Comput. Lang. Syst. Struct.
Maarten Faddegon, Olaf Chitil
2016 conf
IFL
Olaf Chitil, Maarten Faddegon, Colin Runciman
2016 A* conf
PLDI
Maarten Faddegon, Olaf Chitil
2015 A* conf
PLDI
Maarten Faddegon, Olaf Chitil
2014 C ed.
PPDP
Olaf Chitil, Andy King, Olivier Danvy
2014 conf
Trends in Functional Programming
Maarten Faddegon, Olaf Chitil
2012 A conf
ICFP
Olaf Chitil
2011 C conf
PEPM
Olaf Chitil
2011 ed.
IFL
Sven-Bodo Scholz, Olaf Chitil
2009 J jnl
J. Funct. Program.
S. Doaitse Swierstra, Olaf Chitil
2008 C conf
PPDP
Olaf Chitil, Thomas Davie
2008 ch.
Wiley Encyclopedia of Computer Science and Engineering
Olaf Chitil
2008 ed.
IFL
Olaf Chitil, Zoltán Horváth, Viktória Zsók
2007 B conf
APLAS
Olaf Chitil, Frank Huch
2006 conf
IFL
Olaf Chitil, Frank Huch
2006 C conf
PPDP
Josep Silva, Olaf Chitil
2006 conf
Trends in Functional Programming
Yong Luo, Olaf Chitil
2006 conf
TERMGRAPH@ETAPS
Olaf Chitil, Yong Luo
2005 J jnl
ACM Trans. Program. Lang. Syst.
Olaf Chitil
2004 C conf
PADL
Bernd Brassel, Olaf Chitil, Michael Hanus, Frank Huch
2004 conf
IFL
Olaf Chitil
2003 conf
IFL
Olaf Chitil, Dan McNeill, Colin Runciman
2002 conf
Advanced Functional Programming
Koen Claessen, Colin Runciman, Olaf Chitil, John Hughes, Malcolm Wallace
2002 conf
IFL
Olaf Chitil, Colin Runciman, Malcolm Wallace
2001 A conf
ICFP
Olaf Chitil
2000 conf
IFL
Olaf Chitil, Colin Runciman, Malcolm Wallace
2000
Olaf Chitil
1999 A conf
ICFP
Olaf Chitil
1999 conf
IFL
Olaf Chitil
1997 conf
Implementation of Functional Languages
Olaf Chitil
1997 J jnl
Fundam. Informaticae
Olaf Chitil
redb/extractors/js_extractors/js_suspicious_apis.py
← Index redb/extractors/js_extractors/js_suspicious_apis.py python
import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor
from redb.extractors.js_extractors.js_patterns import CATEGORIES, PATTERNS


# Backwards-compatible export: `{category: [(raw_pattern_string, api_name), ...]}`
# in canonical PATTERNS insertion order (code_execution, network, filesystem,
# process, registry, crypto_encoding, dom_manipulation). Kept so external
# callers (notably JSDeobfuscationExtractor pre-cleanup) keep working until
# they are migrated to PATTERNS directly.
SUSPICIOUS_APIS: "dict[str, list[tuple[str, str]]]" = {}
for _name, _compiled in PATTERNS.items():
    SUSPICIOUS_APIS.setdefault(CATEGORIES[_name], []).append((_compiled.pattern, _name))


class JSSuspiciousAPIsExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.api_findings = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_SUSPICIOUS_APIS.value

    def _get_context_snippet(self, line, max_len=200):
        """Get a truncated context snippet around a match."""
        line = line.strip()
        if len(line) > max_len:
            return line[:max_len] + "..."
        return line

    def extract(self):
        src = self.js_source
        if not src:
            return None

        # Pass 1: shared per-sample scan over the raw source. The dict contains
        # entries for both PATTERNS and FEATURE_PATTERNS; the loop below only
        # consults PATTERNS keys, so feature-only entries are ignored.
        raw_scan = self._context.scan or {}
        raw_lines = self.lines

        # Pass 2: same patterns over the deobfuscated text, when the
        # deobfuscator produced something meaningfully different. APIs hidden
        # behind one obfuscation layer (Vjw0rm-style array.join + eval,
        # Dean-Edwards packers, jjencode, ...) only surface here. The scan is
        # cached on JSContext so JSDeobfuscationExtractor (which computes the
        # new_apis_found diff) reuses the same result.
        deobf_scan = self._context.scan_deobfuscated
        if deobf_scan:
            deobf_text, _ = self._context.deobfuscated
            deobf_lines = deobf_text.splitlines()
        else:
            deobf_lines = []

        findings = []
        # Iterate PATTERNS in canonical order so output is deterministic and
        # matches the historical category/pattern ordering. For each api_name,
        # raw findings take precedence; if an API is found only in the
        # deobfuscated text, we surface it as a row tagged revealed_by_deobf=1
        # with line numbers / snippets pulled from the deobfuscated source.
        for api_name in PATTERNS:
            raw_info = raw_scan.get(api_name)
            if raw_info:
                line_numbers = raw_info["lines"]
                lines_for_snippets = raw_lines
                revealed_by_deobf = 0
            else:
                deobf_info = deobf_scan.get(api_name)
                if not deobf_info:
                    continue
                line_numbers = deobf_info["lines"]
                lines_for_snippets = deobf_lines
                revealed_by_deobf = 1

            snippets = [
                self._get_context_snippet(lines_for_snippets[ln - 1])
                for ln in line_numbers[:3]
                if 0 < ln <= len(lines_for_snippets)
            ]
            findings.append({
                "api_name": api_name,
                "api_category": CATEGORIES[api_name],
                # Historical semantics: count = number of unique lines with a
                # match, not total in-source match count.
                "call_count": len(line_numbers),
                "line_numbers": line_numbers,
                "context_snippet": " | ".join(snippets),
                "revealed_by_deobf": revealed_by_deobf,
            })

        if not findings:
            return None

        self.api_findings = findings
        return findings

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type == "ClickHouseExporter":
            if not self.api_findings:
                return None

            current_time = datetime.now(timezone.utc)
            data = []
            for f in self.api_findings:
                data.append([
                    self.sha256,
                    f['api_name'],
                    f['api_category'],
                    f['call_count'],
                    f['line_numbers'],
                    f['context_snippet'],
                    f['revealed_by_deobf'],
                    current_time,
                ])

            column_names = [
                "sha256", "api_name", "api_category",
                "call_count", "line_numbers", "context_snippet",
                "revealed_by_deobf",
                "analysis_date",
            ]

            column_type_names = [
                "FixedString(64)", "String", "LowCardinality(String)",
                "UInt32", "Array(UInt32)", "String",
                "UInt8",
                "DateTime64(3, 'UTC')",
            ]

            return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "redb_js_suspicious_apis"