Valentin Zacharias

34 papers A 2B 2C 1Misc 2Journal 2Unranked 22
YearRankTypeTitle / Venue / Authors
2014 conf
MKWI
Stefan Hellfeld, Jeron Mehl, Andreas Oberweis, Valentin Zacharias
2014 ch.
Towards the Internet of Services
Rudi Studer, Catherina Burghart, Nenad Stojanovic, Thanh Tran, Valentin Zacharias
2013 B conf
EC-TEL
Verónica Rivera-Pelayo, Emanuel Lacic, Valentin Zacharias, Rudi Studer
2013 A conf
LAK
Verónica Rivera-Pelayo, Johannes Munk, Valentin Zacharias, Simone Braun
2012 A conf
LAK
Verónica Rivera-Pelayo, Valentin Zacharias, Lars Müller, Simone Braun
2012 conf
EnviroInfo
Andreas Abecker, Simone Braun, Wassilios Kazakos, Valentin Zacharias
2011 J jnl
Int. J. Knowl. Eng. Data Min.
Athanasios Mazarakis, Simone Braun, Valentin Zacharias
2011 conf
Foundations for the Web of Information and Services
Andreas Eberhart, Peter Haase, Daniel Oberle, Valentin Zacharias
2011 conf
Foundations for the Web of Information and Services
Andreas Abecker, Ernst Biesalski, Simone Braun, Mark Hefke, Valentin Zacharias
2010 Misc conf
MuC
Simone Braun, Andreas P. Schmidt, Valentin Zacharias
2010 B conf
EKAW
Maryam Ramezani, Hans Friedrich Witschel, Simone Braun, Valentin Zacharias
2009 J jnl
i-com
Simone Braun, Andreas P. Schmidt, Valentin Zacharias
2009 conf
I-SEMANTICS
Simone Braun, Claudiu Schora, Valentin Zacharias
2008 conf
RuleML
Valentin Zacharias
2008 conf
ICEIS (2)
Valentin Zacharias
2008 conf
PAKM
Simone Braun, Valentin Zacharias, Hans-Jörg Happel
2008 conf
ICEIS
Valentin Zacharias
2008
Valentin Zacharias
2008 conf
OTM Conferences (2)
Simone Braun, Andreas P. Schmidt, Andreas Walter, Valentin Zacharias
2007 conf
SFSW
Valentin Zacharias, Andreas Abecker
2007 conf
New Forms of Reasoning for the Semantic Web
Valentin Zacharias, Andreas Abecker, Denny Vrandecic, Imen Borgi, Simone Braun, Andreas P. Schmidt
2007 C conf
SEKE
Valentin Zacharias, Andreas Abecker
2007 conf
CKC
Simone Braun, Andreas P. Schmidt, Andreas Walter, Gábor Nagypál, Valentin Zacharias
2007 conf
CKC
Valentin Zacharias, Simone Braun
2007 Misc conf
MuC
Simone Braun, Andreas P. Schmidt, Valentin Zacharias
2007 conf
ISD (1)
Valentin Zacharias
2007 conf
ESOE
Simone Braun, Andreas P. Schmidt, Andreas Walter, Valentin Zacharias
2006 conf
ICEIS (2)
Mark Hefke, Valentin Zacharias, Andreas Abecker, Qingli Wang, Ernst Biesalski, Marco Breiter
2006 conf
SAAW@ISWC
Valentin Zacharias
2005 conf
CIMCA/IAWTIC
Valentin Zacharias
2004 conf
LWA
Valentin Zacharias, Mike Sibler
2003 ch.
Text Mining
Andreas Hotho, Alexander Maedche, Steffen Staab, Valentin Zacharias
2002 conf
PKDD
Alexander Maedche, Valentin Zacharias
2002 conf
EC-Web
Erol Bozsak, Marc Ehrig, Siegfried Handschuh, Andreas Hotho, Alexander Maedche, Boris Motik, Daniel Oberle, Christoph Schmitz, Steffen Staab, Ljiljana Stojanovic, Nenad Stojanovic, Rudi Studer, Gerd Stumme, York Sure, Julien Tane, Raphael Volz, Valentin Zacharias
redb/extractors/js_extractors/js_suspicious_apis.py
← Index redb/extractors/js_extractors/js_suspicious_apis.py python
import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor
from redb.extractors.js_extractors.js_patterns import CATEGORIES, PATTERNS


# Backwards-compatible export: `{category: [(raw_pattern_string, api_name), ...]}`
# in canonical PATTERNS insertion order (code_execution, network, filesystem,
# process, registry, crypto_encoding, dom_manipulation). Kept so external
# callers (notably JSDeobfuscationExtractor pre-cleanup) keep working until
# they are migrated to PATTERNS directly.
SUSPICIOUS_APIS: "dict[str, list[tuple[str, str]]]" = {}
for _name, _compiled in PATTERNS.items():
    SUSPICIOUS_APIS.setdefault(CATEGORIES[_name], []).append((_compiled.pattern, _name))


class JSSuspiciousAPIsExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.api_findings = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_SUSPICIOUS_APIS.value

    def _get_context_snippet(self, line, max_len=200):
        """Get a truncated context snippet around a match."""
        line = line.strip()
        if len(line) > max_len:
            return line[:max_len] + "..."
        return line

    def extract(self):
        src = self.js_source
        if not src:
            return None

        # Pass 1: shared per-sample scan over the raw source. The dict contains
        # entries for both PATTERNS and FEATURE_PATTERNS; the loop below only
        # consults PATTERNS keys, so feature-only entries are ignored.
        raw_scan = self._context.scan or {}
        raw_lines = self.lines

        # Pass 2: same patterns over the deobfuscated text, when the
        # deobfuscator produced something meaningfully different. APIs hidden
        # behind one obfuscation layer (Vjw0rm-style array.join + eval,
        # Dean-Edwards packers, jjencode, ...) only surface here. The scan is
        # cached on JSContext so JSDeobfuscationExtractor (which computes the
        # new_apis_found diff) reuses the same result.
        deobf_scan = self._context.scan_deobfuscated
        if deobf_scan:
            deobf_text, _ = self._context.deobfuscated
            deobf_lines = deobf_text.splitlines()
        else:
            deobf_lines = []

        findings = []
        # Iterate PATTERNS in canonical order so output is deterministic and
        # matches the historical category/pattern ordering. For each api_name,
        # raw findings take precedence; if an API is found only in the
        # deobfuscated text, we surface it as a row tagged revealed_by_deobf=1
        # with line numbers / snippets pulled from the deobfuscated source.
        for api_name in PATTERNS:
            raw_info = raw_scan.get(api_name)
            if raw_info:
                line_numbers = raw_info["lines"]
                lines_for_snippets = raw_lines
                revealed_by_deobf = 0
            else:
                deobf_info = deobf_scan.get(api_name)
                if not deobf_info:
                    continue
                line_numbers = deobf_info["lines"]
                lines_for_snippets = deobf_lines
                revealed_by_deobf = 1

            snippets = [
                self._get_context_snippet(lines_for_snippets[ln - 1])
                for ln in line_numbers[:3]
                if 0 < ln <= len(lines_for_snippets)
            ]
            findings.append({
                "api_name": api_name,
                "api_category": CATEGORIES[api_name],
                # Historical semantics: count = number of unique lines with a
                # match, not total in-source match count.
                "call_count": len(line_numbers),
                "line_numbers": line_numbers,
                "context_snippet": " | ".join(snippets),
                "revealed_by_deobf": revealed_by_deobf,
            })

        if not findings:
            return None

        self.api_findings = findings
        return findings

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type == "ClickHouseExporter":
            if not self.api_findings:
                return None

            current_time = datetime.now(timezone.utc)
            data = []
            for f in self.api_findings:
                data.append([
                    self.sha256,
                    f['api_name'],
                    f['api_category'],
                    f['call_count'],
                    f['line_numbers'],
                    f['context_snippet'],
                    f['revealed_by_deobf'],
                    current_time,
                ])

            column_names = [
                "sha256", "api_name", "api_category",
                "call_count", "line_numbers", "context_snippet",
                "revealed_by_deobf",
                "analysis_date",
            ]

            column_type_names = [
                "FixedString(64)", "String", "LowCardinality(String)",
                "UInt32", "Array(UInt32)", "String",
                "UInt8",
                "DateTime64(3, 'UTC')",
            ]

            return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "redb_js_suspicious_apis"