Victoria Vesna

19 papers A* 2Journal 11Unranked 4
YearRankTypeTitle / Venue / Authors
2024 J jnl
AI Soc.
Victoria Vesna
2020 J jnl
AI Soc.
John Brumley, Charles E. Taylor, Reiji Suzuki, Takashi Ikegami, Victoria Vesna, Hiroo Iwata
2019 J jnl
IEEE Computer Graphics and Applications
Victoria Vesna, Bruce Donald Campbell, Francesca Samsel
2018 J jnl
AI Soc.
Aisen Caro Chacin, Hiroo Iwata, Victoria Vesna
2016 ch.
Handbook of Science and Technology Convergence
James K. Gimzewski, Adam Z. Stieg, Victoria Vesna
2012 J jnl
AI Soc.
Victoria Vesna
2010 A* conf
ACM Multimedia
Victoria Vesna, James K. Gimzewski
2009 conf
SIGGRAPH Art Gallery
Victoria Vesna, W. H. Lucas, Claes Andersson, Jay Yan
2009 conf
SIGGRAPH Art Gallery
Michael Kelly, Victoria Vesna, Paul A. Fishwick, Andrew Vande Moere, Kenneth A. Huff
2008 ch.
Cognition, Communication and Interaction
Victoria Vesna
2006 J jnl
AI Soc.
Victoria Vesna
2006 J jnl
AI Soc.
Victoria Vesna
2004 conf
SIGGRAPH Art Gallery
Roy Ascott, Donna J. Cox, Margaret Dolinsky, Diane Gromala, Marcos Novak, Miroslaw Rogala, Thecla Schiphorst, Diana Slattery, Victoria Vesna
2000 J jnl
AI Soc.
Victoria Vesna
2000 J jnl
AI Soc.
Victoria Vesna
2000 J jnl
AI Soc.
Victoria Vesna
1998 J jnl
Digit. Creativity
Victoria Vesna
1996 conf
SIGGRAPH Visual Proceedings
Victoria Vesna
1996 A* conf
SIGGRAPH
Perry Hoberman, Victoria Vesna
redb/extractors/decompiler/bninja/similarity/minhasher.py
← Index redb/extractors/decompiler/bninja/similarity/minhasher.py python
import logging
import random
from enum import Enum

from ..analysis.medium_level_normalization import MediumLevelNormalization

try:
    from .minhashcustom import MinHashCustom
    from ..analysis.low_level_normalization import LowLevelNormalization
except ImportError:
    # Fallback to absolute imports (for multiprocessing spawned processes)
    from redb.extractors.decompiler.bninja.similarity.minhashcustom import MinHashCustom
    from redb.extractors.decompiler.bninja.analysis.low_level_normalization import LowLevelNormalization

## Values for this configuration were extracted from https://github.com/danielplohmann/mcrit/blob/main/mcrit/config/MinHashConfig.py#L10
# Length in number of Shingles of which a minhash consists
# this value represents the length of sha256sum hash truncated
MINHASH_SIGNATURE_LENGTH: int = 64
# Number of bits per signature element (1-32 bits)
MINHASH_SIGNATURE_BITS: int = 8


class TokenKind(Enum):
    LLIL = "llil"
    TYPED_LLIL = "typed_llil"
    MLIL = "mlil"
    TYPED_MLIL = "typed_mlil"


class MinHasher:
    # stick to the default method
    MINHASH_STRATEGY_HASH_ALL = 1

    def __init__(self, seed, il_function, kind: TokenKind = TokenKind.LLIL):
        self._minhash_seeds = []
        self.il_func = il_function
        self.kind = kind
        self._minhash_permutation = []
        self._signature_segments = []
        self._initMinhashing(seed)

    def _initMinhashing(self, MINHASH_SEED=None):
        random.seed(MINHASH_SEED)
        # init sequence of seeds
        self._minhash_seeds = [
            random.randint(0, MinHashCustom.getHashMax()) for _ in range(MINHASH_SIGNATURE_LENGTH)
        ]

    def make_ngrams(self, tokens, n=3):
        """Take the ngrams of the IL we try to pass into the functions"""
        return [tuple(tokens[i:i+n]) for i in range(len(tokens) - n + 1)]

    def _extract_tokens(self):
        """Extract the IL tokens from the IL function, picking the right
        normalizer (LLIL/MLIL) and the right normalization mode
        (skeleton/typed) based on self.kind."""
        if self.kind in (TokenKind.LLIL, TokenKind.TYPED_LLIL):
            normalizer = LowLevelNormalization()
        elif self.kind in (TokenKind.MLIL, TokenKind.TYPED_MLIL):
            normalizer = MediumLevelNormalization()
        else:
            raise ValueError(f"Unsupported token kind: {self.kind}")

        # typed variants include operand type info, skeleton variants don't
        if self.kind in (TokenKind.TYPED_LLIL, TokenKind.TYPED_MLIL):
            normalize = normalizer.normalize_instr_with_operands
        else:
            normalize = normalizer.normalize_instruction_all_levels

        instructions = []
        for basic_block in self.il_func.basic_blocks:
            for il in basic_block:
                instructions.append(normalize(il))

        return instructions

    def calculateMinHash(self):
        """Calculate hash function every time, then take minimum shingle per shingler"""
        minhash_result = MinHashCustom(minhash_bits=MINHASH_SIGNATURE_BITS)
        minhash_signature = []

        tokens = self._extract_tokens()
        shingles = self.make_ngrams(tokens, n=3)

        # Functions with fewer than 3 IL instructions can't produce n-grams
        # Return empty minhash for such small functions (thunks, stubs, etc.)
        # Triggered by 39d8ad95b0323c37bd3134ab93ac4af44c66a1a8443a41c1ac02cec19bb2816a
        if not shingles:
            return []

        # Generate the MinHash
        for seed in self._minhash_seeds:
            hashed_shingles = [
                self.shingle_hash(shingle, seed) for shingle in shingles
            ]
            min_value = min(hashed_shingles)

            if MINHASH_SIGNATURE_BITS < 32:
                min_value %= (2 ** MINHASH_SIGNATURE_BITS)

            minhash_signature.append(min_value)

        minhash_result.setMinHash(minhash_signature)
        return minhash_result.getMinHashInt()

    def shingle_hash(self, shingle, hash_seed=0):
        """Produce a single 32bit UINT hash for a given shingle"""
        return MinHashCustom.hashData(shingle, hash_seed)