Jakob Fehle

16 papers B 1Misc 2Journal 9Unranked 4
YearRankTypeTitle / Venue / Authors
2026 J jnl
CoRR
Nils Constantin Hellwig, Jakob Fehle, Udo Kruschwitz, Christian Wolff
2026 J jnl
CoRR
Nils Constantin Hellwig, Jakob Fehle, Udo Kruschwitz, Christian Wolff
2026 J jnl
Knowl. Based Syst.
Jakob Fehle, Udo Kruschwitz, Nils Constantin Hellwig, Christian Wolff
2026 J jnl
CoRR
Nils Constantin Hellwig, Jakob Fehle, Udo Kruschwitz, Christian Wolff
2025 J jnl
CoRR
Nils Constantin Hellwig, Jakob Fehle, Udo Kruschwitz, Christian Wolff
2025 J jnl
Expert Syst. Appl.
Nils Constantin Hellwig, Jakob Fehle, Christian Wolff
2024 J jnl
CoRR
Michael Achmann-Denkler, Jakob Fehle, Mario Haim, Christian Wolff
2024 J jnl
Int. J. Speech Technol.
Nils Constantin Hellwig, Jakob Fehle, Markus Bink, Thomas Schmidt, Christian Wolff
2024 conf
KONVENS
Nils Constantin Hellwig, Jakob Fehle, Markus Bink, Christian Wolff
2024 J jnl
CoRR
Nils Constantin Hellwig, Jakob Fehle, Markus Bink, Christian Wolff
2023 conf
KONVENS
Jakob Fehle, Leonie Münster, Thomas Schmidt, Christian Wolff
2023 conf
ICNLSP
Nils Constantin Hellwig, Markus Bink, Thomas Schmidt, Jakob Fehle, Christian Wolff
2022 conf
KONVENS
Thomas Schmidt, Jakob Fehle, Maximilian Weissenbacher, Jonathan Richter, Philipp Gottschalk, Christian Wolff
2022 Misc conf
MuC
David Halbhuber, Maximilian Seewald, Fabian Schiller, Mathias Götz, Jakob Fehle, Niels Henze
2020 B conf
VRST
Valentin Schwind, David Halbhuber, Jakob Fehle, Jonathan Sasse, Andreas Pfaffelhuber, Christoph Tögel, Julian Dietz, Niels Henze
2019 Misc conf
MuC
David Halbhuber, Jakob Fehle, Alexander Kalus, Konstantin Seitz, Martin Kocur, Thomas Schmidt, Christian Wolff
tests/unit/test_decompile_strings.py
← Index tests/unit/test_decompile_strings.py python
"""Unit tests for bninja/analysis/strings.py — StringAnalysis."""
import pytest
import math
from unittest.mock import MagicMock


# StringAnalysis has no binaryninja imports, just collections and math
from redb.extractors.decompiler.bninja.analysis.strings import StringAnalysis


# ============================================================================
# Helper mocks
# ============================================================================

class MockStringEntry:
    """Mock for a Binary Ninja string reference."""
    def __init__(self, value, raw=None, start=0, length=0, type_name="Utf8String"):
        self.value = value
        self.raw = raw if raw is not None else (value.encode("utf-8") if isinstance(value, str) else value)
        self.start = start
        self.length = length if length else len(self.raw)
        self.type = MagicMock()
        self.type.name = type_name


class MockBinaryView:
    """Mock binary view with a strings list."""
    def __init__(self, strings=None):
        self.strings = strings or []


# ============================================================================
# 6a. StringAnalysis
# ============================================================================


class TestStringAnalysisEntropy:
    def setup_method(self):
        self.sa = StringAnalysis(bv=MockBinaryView(), functions=[])

    def test_entropy_empty_string(self):
        assert self.sa.entropy("") == 0.0

    def test_entropy_single_char(self):
        assert self.sa.entropy("aaaa") == 0.0

    def test_entropy_uniform_distribution(self):
        # "abcd" -> 4 unique chars, each p=1/4, entropy = log2(4) = 2.0
        result = self.sa.entropy("abcd")
        assert result == pytest.approx(2.0)

    def test_entropy_binary_string(self):
        # "ab" -> 2 unique chars, each p=1/2, entropy = log2(2) = 1.0
        result = self.sa.entropy("ab")
        assert result == pytest.approx(1.0)


class TestStringAnalysisAnalyze:
    def test_analyze_deduplication(self):
        """Duplicate (string, encoding) pairs -> only first kept."""
        entries = [
            MockStringEntry("hello", start=100, type_name="Utf8String"),
            MockStringEntry("hello", start=200, type_name="Utf8String"),
        ]
        bv = MockBinaryView(strings=entries)
        sa = StringAnalysis(bv=bv, functions=[])
        result = sa.analyze()
        assert len(result) == 1
        assert result[0]["string_offset"] == 100

    def test_analyze_sorted_by_address(self):
        """First occurrence (lowest offset) is the one kept."""
        entries = [
            MockStringEntry("world", start=500, type_name="Utf8String"),
            MockStringEntry("world", start=100, type_name="Utf8String"),
        ]
        bv = MockBinaryView(strings=entries)
        sa = StringAnalysis(bv=bv, functions=[])
        result = sa.analyze()
        assert len(result) == 1
        # The analyze() sorts by start, so 100 comes first
        assert result[0]["string_offset"] == 100

    def test_analyze_empty_bv(self):
        bv = MockBinaryView(strings=[])
        sa = StringAnalysis(bv=bv, functions=[])
        result = sa.analyze()
        assert result == []

    def test_analyze_output_schema(self):
        entries = [MockStringEntry("test_string", start=0, type_name="Utf8String")]
        bv = MockBinaryView(strings=entries)
        sa = StringAnalysis(bv=bv, functions=[])
        result = sa.analyze()
        assert len(result) == 1
        r = result[0]
        required_keys = [
            "string",
            "string_raw",
            "string_encoding",
            "string_offset",
            "string_length",
            "string_raw_length",
            "string_entropy",
        ]
        for key in required_keys:
            assert key in r, f"Missing key: {key}"