Vasily Volkov

16 papers A 3B 1Misc 1Journal 4Unranked 5
YearRankTypeTitle / Venue / Authors
2018 B conf
PPoPP
Vasily Volkov
2016 J jnl
Concurr. Comput. Pract. Exp.
Adnan Salman, Allen D. Malony, Sergei Turovets, Vasily Volkov, David Ozog, Don M. Tucker
2016
Vasily Volkov
2014 J jnl
Comput. Math. Methods Medicine
Sergei Turovets, Vasily Volkov, Aleksei Zherdetsky, Alena Prakonina, Allen D. Malony
2013 conf
HPCS
Adnan Salman, Allen D. Malony, Sergei Turovets, Vasily Volkov, David Ozog, Don M. Tucker
2011 conf
MMVR
Allen D. Malony, Adnan Salman, Sergei Turovets, Don M. Tucker, Vasily Volkov, Kai Li, Jung Eun Song, Scott Biersdorff, Colin Davey, Chris Hoge, David K. Hammond
2010 ch.
Scientific Computing with Multicore and Accelerators
Kaushik Datta, Samuel Williams, Vasily Volkov, Jonathan Carter, Leonid Oliker, John Shalf, Katherine A. Yelick
2009 conf
ICCS (1)
Vasily Volkov, Aleksei Zherdetsky, Sergei Turovets, Allen D. Malony
2008 A conf
SC
Vasily Volkov, James Demmel
2008 J jnl
IEEE Micro
Michael Garland, Scott Le Grand, John Nickolls, Joshua Anderson, Jim Hardwick, Scott Morton, Everett H. Phillips, Yao Zhang, Vasily Volkov
2008 A conf
SC
Kaushik Datta, Mark Murphy, Vasily Volkov, Samuel Williams, Jonathan Carter, Leonid Oliker, David A. Patterson, John Shalf, Katherine A. Yelick
2006 J jnl
J. Comput. Sci. Technol.
Ling Li, Vasily Volkov
2005 conf
ACSC
Ling Li, Vasily Volkov
2005 conf
IWOMP
Adnan Salman, Sergei Turovets, Allen D. Malony, Vasily Volkov
2004 Misc conf
ICCVG
Vasily Volkov, Ling Li
2003 A conf
IEEE Visualization
Vasily Volkov, Ling Li
redb/extractors/js_extractors/js_deobfuscation.py
← Index redb/extractors/js_extractors/js_deobfuscation.py python
import hashlib
import inspect
import re
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor
from redb.extractors.js_extractors.js_patterns import PATTERNS

# String literals of 4+ characters; only used by the deobfuscation diff to count
# strings revealed after deobfuscation. Compiled once at module load.
_STRING_LITERAL_4PLUS_RE = re.compile(r"[\"\']([^\"\']{4,})[\"\']")


class JSDeobfuscationExtractor(JSExtractor):
    """Compute pre/post-deobfuscation metrics for a JS sample.

    The actual deobfuscation pass (external tool with jsbeautifier fallback)
    lives on `JSContext.deobfuscated` and is cached per sample, so any other
    extractor that needs the deobfuscated text reads the same value without
    re-running the subprocess. Configure the external tool via env vars:
        JS_DEOBFUSCATOR_PATH    Path or name (default: webcrack)
        JS_DEOBFUSCATE_TIMEOUT  Seconds (default: 60)
    """

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.deobfuscation_result = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_DEOBFUSCATION.value

    def extract(self):
        src = self.js_source
        if not src:
            return None

        deobfuscated, deobfuscator_used = self._context.deobfuscated
        if deobfuscated is None:
            return None

        original_size = len(src)
        original_entropy = self._context.text_entropy
        deobfuscated_size = len(deobfuscated)
        deobfuscated_entropy = self._calculate_text_entropy(deobfuscated)
        size_change_ratio = round(deobfuscated_size / original_size, 4) if original_size else 0.0

        # Strings revealed by deobfuscation: matched literals are extracted from
        # both versions and the set difference is the count of "new" strings.
        original_strings = set(_STRING_LITERAL_4PLUS_RE.findall(src))
        deobfuscated_strings = set(_STRING_LITERAL_4PLUS_RE.findall(deobfuscated))
        new_strings = deobfuscated_strings - original_strings

        # Suspicious APIs revealed by deobfuscation. Both sides of the diff
        # come from JSContext caches: the raw scan is computed once for the
        # whole pipeline; the deobfuscated scan is computed once and reused
        # by JSSuspiciousAPIsExtractor's revealed_by_deobf rows.
        original_apis = {n for n in self._context.scan if n in PATTERNS}
        deobfuscated_apis = set(self._context.scan_deobfuscated)
        new_apis = deobfuscated_apis - original_apis

        deobfuscated_sha256 = hashlib.sha256(deobfuscated.encode('utf-8')).hexdigest()

        self.deobfuscation_result = {
            'deobfuscator_used': deobfuscator_used,
            'deobfuscation_successful': True,
            'original_size': original_size,
            'deobfuscated_size': deobfuscated_size,
            'size_change_ratio': size_change_ratio,
            'original_entropy': original_entropy,
            'deobfuscated_entropy': deobfuscated_entropy,
            'new_strings_found': len(new_strings),
            'new_apis_found': len(new_apis),
            'deobfuscated_sha256': deobfuscated_sha256,
        }
        return self.deobfuscation_result

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type == "ClickHouseExporter":
            if not self.deobfuscation_result:
                return None

            r = self.deobfuscation_result
            current_time = datetime.now(timezone.utc)
            data = [[
                self.sha256,
                r['deobfuscator_used'],
                int(r['deobfuscation_successful']),
                r['original_size'],
                r['deobfuscated_size'],
                r['size_change_ratio'],
                r['original_entropy'],
                r['deobfuscated_entropy'],
                r['new_strings_found'],
                r['new_apis_found'],
                r['deobfuscated_sha256'],
                current_time,
            ]]

            column_names = [
                "sha256", "deobfuscator_used", "deobfuscation_successful",
                "original_size", "deobfuscated_size", "size_change_ratio",
                "original_entropy", "deobfuscated_entropy",
                "new_strings_found", "new_apis_found",
                "deobfuscated_sha256", "analysis_date",
            ]

            column_type_names = [
                "FixedString(64)", "LowCardinality(String)", "UInt8",
                "UInt64", "UInt64", "Float64",
                "Float64", "Float64",
                "UInt32", "UInt32",
                "FixedString(64)", "DateTime64(3, 'UTC')",
            ]

            return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "redb_js_deobfuscation"