Ines Arous

15 papers A* 6Journal 4Unranked 5
YearRankTypeTitle / Venue / Authors
2026 conf
EACL (Findings)
Bonaventure F. P. Dossou, Ines Arous, Audrey Durand, Jackie Chi Kit Cheung
2025 conf
ACL (4)
Bonaventure F. P. Dossou, Ines Arous, Jackie CK Cheung
2024 conf
ACL (1)
Maxime Darrin, Ines Arous, Pablo Piantanida, Jackie Chi Kit Cheung
2024 J jnl
CoRR
Maxime Darrin, Ines Arous, Pablo Piantanida, Jackie C. K. Cheung
2023 conf
EMNLP (Findings)
Zichao Li, Ines Arous, Siva Reddy, Jackie Chi Kit Cheung
2023 J jnl
CoRR
Zichao Li, Ines Arous, Siva Reddy, Jackie C. K. Cheung
2023 A* conf
WWW
Sepideh Mesbah, Ines Arous, Jie Yang, Alessandro Bozzon
2023 A* conf
WWW
Ines Arous, Ljiljana Dolamic, Philippe Cudré-Mauroux
2021 A* conf
AAAI
Ines Arous, Ljiljana Dolamic, Jie Yang, Akansha Bhardwaj, Giuseppe Cuccu, Philippe Cudré-Mauroux
2021 A* conf
WWW
Ines Arous, Jie Yang, Mourad Khayati, Philippe Cudré-Mauroux
2020 J jnl
Proc. VLDB Endow.
Mourad Khayati, Ines Arous, Zakhar Tymchenko, Philippe Cudré-Mauroux
2020 A* conf
WWW
Ines Arous, Jie Yang, Mourad Khayati, Philippe Cudré-Mauroux
2020 conf
IEEE BigData
Zeno Bardelli, Ines Arous, Philippe Cudré-Mauroux, Ljiljana Dolamic
2019 A* conf
ICDE
Ines Arous, Mourad Khayati, Philippe Cudré-Mauroux, Ying Zhang, Martin L. Kersten, Svetlin Stalinlov
2017 J jnl
CoRR
Alessandro Checco, Gianluca Demartini, Alexander Löser, Ines Arous, Mourad Khayati, Matthias Dantone, Richard Koopmanschap, Svetlin Stalinlov, Martin L. Kersten, Ying Zhang
tests/unit/test_decompile_strings.py
← Index tests/unit/test_decompile_strings.py python
"""Unit tests for bninja/analysis/strings.py — StringAnalysis."""
import pytest
import math
from unittest.mock import MagicMock


# StringAnalysis has no binaryninja imports, just collections and math
from redb.extractors.decompiler.bninja.analysis.strings import StringAnalysis


# ============================================================================
# Helper mocks
# ============================================================================

class MockStringEntry:
    """Mock for a Binary Ninja string reference."""
    def __init__(self, value, raw=None, start=0, length=0, type_name="Utf8String"):
        self.value = value
        self.raw = raw if raw is not None else (value.encode("utf-8") if isinstance(value, str) else value)
        self.start = start
        self.length = length if length else len(self.raw)
        self.type = MagicMock()
        self.type.name = type_name


class MockBinaryView:
    """Mock binary view with a strings list."""
    def __init__(self, strings=None):
        self.strings = strings or []


# ============================================================================
# 6a. StringAnalysis
# ============================================================================


class TestStringAnalysisEntropy:
    def setup_method(self):
        self.sa = StringAnalysis(bv=MockBinaryView(), functions=[])

    def test_entropy_empty_string(self):
        assert self.sa.entropy("") == 0.0

    def test_entropy_single_char(self):
        assert self.sa.entropy("aaaa") == 0.0

    def test_entropy_uniform_distribution(self):
        # "abcd" -> 4 unique chars, each p=1/4, entropy = log2(4) = 2.0
        result = self.sa.entropy("abcd")
        assert result == pytest.approx(2.0)

    def test_entropy_binary_string(self):
        # "ab" -> 2 unique chars, each p=1/2, entropy = log2(2) = 1.0
        result = self.sa.entropy("ab")
        assert result == pytest.approx(1.0)


class TestStringAnalysisAnalyze:
    def test_analyze_deduplication(self):
        """Duplicate (string, encoding) pairs -> only first kept."""
        entries = [
            MockStringEntry("hello", start=100, type_name="Utf8String"),
            MockStringEntry("hello", start=200, type_name="Utf8String"),
        ]
        bv = MockBinaryView(strings=entries)
        sa = StringAnalysis(bv=bv, functions=[])
        result = sa.analyze()
        assert len(result) == 1
        assert result[0]["string_offset"] == 100

    def test_analyze_sorted_by_address(self):
        """First occurrence (lowest offset) is the one kept."""
        entries = [
            MockStringEntry("world", start=500, type_name="Utf8String"),
            MockStringEntry("world", start=100, type_name="Utf8String"),
        ]
        bv = MockBinaryView(strings=entries)
        sa = StringAnalysis(bv=bv, functions=[])
        result = sa.analyze()
        assert len(result) == 1
        # The analyze() sorts by start, so 100 comes first
        assert result[0]["string_offset"] == 100

    def test_analyze_empty_bv(self):
        bv = MockBinaryView(strings=[])
        sa = StringAnalysis(bv=bv, functions=[])
        result = sa.analyze()
        assert result == []

    def test_analyze_output_schema(self):
        entries = [MockStringEntry("test_string", start=0, type_name="Utf8String")]
        bv = MockBinaryView(strings=entries)
        sa = StringAnalysis(bv=bv, functions=[])
        result = sa.analyze()
        assert len(result) == 1
        r = result[0]
        required_keys = [
            "string",
            "string_raw",
            "string_encoding",
            "string_offset",
            "string_length",
            "string_raw_length",
            "string_entropy",
        ]
        for key in required_keys:
            assert key in r, f"Missing key: {key}"