Wei Liu

24 papers A* 3B 3C 1Journal 13Unranked 3
YearRankTypeTitle / Venue / Authors
2011 J jnl
IEEE J. Sel. Areas Commun.
Wei Liu, Chi Zhang, Guoliang Yao, Yuguang Fang
2009 J jnl
Wirel. Networks
Wenjing Lou, Wei Liu, Yanchao Zhang, Yuguang Fang
2007 J jnl
Wirel. Networks
Yanchao Zhang, Wenjing Lou, Wei Liu, Yuguang Fang
2006 J jnl
Wirel. Networks
Wei Liu, Yanchao Zhang, Wenjing Lou, Yuguang Fang
2006 J jnl
J. Comb. Optim.
Wei Liu, Yanchao Zhang, Yuguang Fang, Kejie Lu
2006 J jnl
Wirel. Networks
Xiang Chen, Wei Liu, Hongqiang Zhai, Yuguang Fang
2006 J jnl
IEEE J. Sel. Areas Commun.
Yanchao Zhang, Wei Liu, Wenjing Lou, Yuguang Fang
2006 J jnl
IEEE Trans. Wirel. Commun.
Yanchao Zhang, Wei Liu, Wenjing Lou, Yuguang Fang
2006 J jnl
IEEE Trans. Veh. Technol.
Xiaojiang Du, Dapeng Wu, Wei Liu, Yuguang Fang
2006 J jnl
IEEE J. Sel. Areas Commun.
Yanchao Zhang, Wei Liu, Yuguang Fang, Dapeng Wu
2006 J jnl
IEEE Trans. Dependable Secur. Comput.
Yanchao Zhang, Wei Liu, Wenjing Lou, Yuguang Fang
2005 conf
ICC
Yanchao Zhang, Wei Liu, Wenjing Lou, Yuguang Fang, Younggoo Kwon
2005 J jnl
Comput. Networks
Wei Liu, Wenjing Lou, Yuguang Fang
2005 A* conf
INFOCOM
Yanchao Zhang, Wei Liu, Wenjing Lou
2005 ch.
Handbook of Algorithms for Wireless Networking and Mobile Computing
Xiang Chen, Wei Liu, Yuguang Fang
2005 B conf
WCNC
Yanchao Zhang, Wei Liu, Wenjing Lou, Yuguang Fang
2004 A* conf
INFOCOM
Wei Liu, Yuguang Fang
2004 J jnl
IEEE Trans. Mob. Comput.
Wei Liu, Xiang Chen, Yuguang Fang, John M. Shea
2004 C conf
QSHINE
Wei Liu, Yanchao Zhang, Wenjing Lou, Yuguang Fang
2004 A* conf
INFOCOM
Wenjing Lou, Wei Liu, Yuguang Fang
2004 B conf
GLOBECOM
Wei Liu, Yanchao Zhang, Wenjing Lou, Yuguang Fang, Tan F. Wong
2004 conf
ICC
Xiang Chen, Wei Liu, Yuguang Fang, Maria C. Yuang
2003 B conf
GLOBECOM
Wei Liu, Wenjing Lou, Xiang Chen, Yuguang Fang
2003 conf
ICC
Wei Liu, Wenjing Lou, Yuguang Fang
redb/extractors/ioc_extractor/ioc_extractor.py
← Index redb/extractors/ioc_extractor/ioc_extractor.py python
"""
IOC Extractor - Extractor class for extracting IOCs from decompilation results.

This extractor works with in-memory data from DecompileBinja, following the
standard Extractor pattern to support both ClickHouse and PrintExporter (dry-run).

Usage:
    # After DecompileBinja completes:
    ioc_extractor = IOCExtractorFromResults(
        analysis_results=decompiler.analysis_results,
        sha256=sha256,
        log=logger,
        exporters=exporters,
        index_prefix=index_prefix
    )
    ioc_extractor.export_data()
"""

import inspect
from datetime import datetime, timezone
from pathlib import Path
from typing import Any, List, Dict, Optional

from redb.extractors.enum import Tag
from redb.extractors.database_exporters import DatabaseExporter

# Import the IOCScraper and related classes from standalone module
from redb.extractors.ioc_extractor.standalone_ioc_extractor import (
    IOCScraper,
    IOCType,
    SourceType,
    ExtractedIOC,
)
from typing import Set


class IOCExtractorFromResults:
    """
    Extracts IOCs from in-memory decompilation results.

    This follows a simplified Extractor pattern but doesn't inherit from Extractor
    since it doesn't read from a binary file - instead it takes already-processed
    analysis results from DecompileBinja.
    """

    def __init__(
        self,
        analysis_results: Dict[str, Any],
        sha256: str,
        log: Any,
        exporters: Optional[List[DatabaseExporter]] = None,
        index_prefix: Optional[str] = None,
        tld_file: Optional[Path] = None,
        suppress_types: Optional[Set[IOCType]] = None,
        js_context: bool = False,
    ):
        """
        Initialize IOC Extractor with analysis results.

        Args:
            analysis_results: Dict containing 'strings' and 'decompiled' lists from DecompileBinja
            sha256: Sample SHA256 hash
            log: Logger instance
            exporters: List of database exporters (ClickHouse, Print, etc.)
            index_prefix: Index prefix for database
            tld_file: Optional path to TLD list file
            js_context: When True, the underlying IOCScraper rejects FQDN
                candidates that match JS object-access syntax (see
                JS_FP_TLDS / JS_FP_SLDS). Set this for the JS pipeline only;
                APK suppresses FQDN entirely via suppress_types and binary
                callers leave it disabled.
        """
        self.log = log
        self.log.debug(f"Creating {self.__class__.__name__}")
        self.analysis_results = analysis_results
        self.sha256 = sha256
        self.exporters = exporters or []
        self.index_prefix = index_prefix
        self.scraper = IOCScraper(
            tld_file, suppress_types=suppress_types, js_context=js_context,
        )
        self.extracted_iocs: List[ExtractedIOC] = []

    def extract(self) -> List[ExtractedIOC]:
        """
        Extract IOCs from strings and decompiled functions in analysis_results.

        Returns:
            List of ExtractedIOC objects
        """
        self.log.debug(inspect.currentframe().f_code.co_name)
        self.extracted_iocs = []

        # Extract from strings
        strings_count = self._extract_from_strings()

        # Extract from decompiled functions
        functions_count = self._extract_from_decompiled()

        # Extract from text-based artefact surfaces (JS, PowerShell, etc.)
        text_count = self._extract_from_text()

        self.log.info(
            f"Extracted {len(self.extracted_iocs)} IOCs for {self.sha256[:16]}... "
            f"(strings: {strings_count}, functions: {functions_count}, "
            f"text: {text_count})"
        )

        return self.extracted_iocs

    def _extract_from_strings(self) -> int:
        """Extract IOCs from sample's strings."""
        count = 0
        strings = self.analysis_results.get("strings", [])

        for s in strings:
            string_value = s.get("string", "")
            string_offset = s.get("string_offset", 0)

            if isinstance(string_value, bytes):
                string_value = string_value.decode('utf-8', errors='replace')

            for ioc in self.scraper.scrape(string_value, SourceType.STRING, str(string_offset)):
                self.extracted_iocs.append(ioc)
                count += 1

        return count

    def _extract_from_decompiled(self) -> int:
        """Extract IOCs from sample's decompiled functions.

        Supports both Binja format (key: "decompiled", fields: "decompiled_function",
        "decompiled_function_hash", "function_type") and APK format (key:
        "decompiled_content", fields: "decompiled_method", "decompiled_method_hash",
        "method_type").
        """
        count = 0

        # Binja format
        decompiled = self.analysis_results.get("decompiled", [])
        for func in decompiled:
            func_type = func.get("function_type", "UNKNOWN")
            if func_type in ("LIBRARY", "THUNK"):
                continue

            func_content = func.get("decompiled_function", "")
            func_hash = func.get("decompiled_function_hash", "unknown")

            if isinstance(func_content, bytes):
                func_content = func_content.decode('utf-8', errors='replace')

            for ioc in self.scraper.scrape(func_content, SourceType.DECOMPILED_FUNCTION, func_hash):
                self.extracted_iocs.append(ioc)
                count += 1

        # APK format (decompiled_content with method-level fields)
        decompiled_content = self.analysis_results.get("decompiled_content", [])
        for func in decompiled_content:
            func_type = func.get("method_type", "UNKNOWN")
            if func_type in ("LIBRARY", "THUNK"):
                continue

            func_content = func.get("decompiled_method", "")
            func_hash = func.get("decompiled_method_hash", "unknown")

            if isinstance(func_content, bytes):
                func_content = func_content.decode('utf-8', errors='replace')

            for ioc in self.scraper.scrape(func_content, SourceType.DECOMPILED_FUNCTION, func_hash):
                self.extracted_iocs.append(ioc)
                count += 1

        return count

    def _extract_from_text(self) -> int:
        """Extract IOCs from text-based artefact surfaces.

        Walks `analysis_results["text_raw"]` and `analysis_results["text_normalized"]`,
        each a list of `{"content": str, "content_hash": str}` dicts. Each
        list is routed through its own SourceType (`TEXT_RAW` /
        `TEXT_NORMALIZED`) so analysts can distinguish IOCs that were already
        present in the raw source from those exposed only after normalisation
        (deobfuscation/beautification). Generic across text-based formats —
        used by JS today, intended for PowerShell, Python, email body,
        extracted PDF/Office text in the future.
        """
        count = 0

        for key, source_type in (
            ("text_raw", SourceType.TEXT_RAW),
            ("text_normalized", SourceType.TEXT_NORMALIZED),
        ):
            for entry in self.analysis_results.get(key, []):
                content = entry.get("content", "")
                content_hash = entry.get("content_hash", "unknown")

                if isinstance(content, bytes):
                    content = content.decode('utf-8', errors='replace')

                for ioc in self.scraper.scrape(content, source_type, content_hash):
                    self.extracted_iocs.append(ioc)
                    count += 1

        return count

    def prepare_export_data(self, exporter_type: str) -> Any:
        """
        Prepare data for specific export type.

        Returns tuple for ClickHouse or list of dicts for Print/Elasticsearch.
        """
        self.log.debug(inspect.currentframe().f_code.co_name)

        if not self.extracted_iocs:
            return None

        now = datetime.now(timezone.utc)

        if exporter_type == "ClickHouseExporter":
            data = [
                [
                    self.sha256,
                    ioc.ioc_type.value,
                    ioc.ioc_value,
                    ioc.source_type.value,
                    ioc.source_identifier,
                    now,
                ]
                for ioc in self.extracted_iocs
            ]

            column_names = [
                "sha256",
                "ioc_type",
                "ioc_value",
                "source_type",
                "source_identifier",
                "extracted_at",
            ]

            column_type_names = [
                "FixedString(64)",
                "Enum8('ipv4'=1, 'ipv6'=2, 'fqdn'=3, 'url'=4, 'email'=5, 'server'=6, "
                "'hash_md5'=10, 'hash_sha1'=11, 'hash_sha256'=12, 'cve'=20, 'cwe'=21, 'cpe'=22, "
                "'crypto_btc'=30, 'crypto_eth'=31, 'crypto_xrp'=32, 'crypto_bch'=33, "
                "'crypto_ada'=34, 'crypto_substrate'=35, 'path_linux'=40, 'path_windows'=41, "
                "'registry_key'=42, 'onion'=50)",
                "String",
                "Enum8('decompiled_function'=1, 'disassembled_function'=2, 'string'=3, "
                "'text_raw'=4, 'text_normalized'=5)",
                "String",
                "DateTime64(3, 'UTC')",
            ]

            return (data, column_names, column_type_names)

        else:
            # For PrintExporter and others - return list of dicts
            return [
                {
                    "sha256": self.sha256,
                    "ioc_type": ioc.ioc_type.value,
                    "ioc_value": ioc.ioc_value,
                    "source_type": ioc.source_type.value,
                    "source_identifier": ioc.source_identifier,
                    "extracted_at": now.isoformat(),
                }
                for ioc in self.extracted_iocs
            ]

    def get_clickhouse_table(self) -> str:
        """Return the ClickHouse table name for IOCs."""
        return "redb_iocs"

    def tag(self) -> str:
        """Return the tag for this extractor."""
        return Tag.IOC.value if hasattr(Tag, 'IOC') else "ioc"

    def export_data(self) -> bool:
        """
        Export extracted IOCs to all configured exporters.

        Returns:
            True if export succeeded, False if failed, None if no data
        """
        self.log.debug(inspect.currentframe().f_code.co_name)

        # First extract the IOCs
        extracted = self.extract()

        if not extracted:
            self.log.debug("No IOCs extracted, skipping export")
            return None

        success = True

        from redb.extractors.database_exporters import PrintExporter, ClickHouseExporter

        for exporter in self.exporters:
            try:
                if isinstance(exporter, PrintExporter):
                    # For PrintExporter, pass the list of dicts
                    export_data = self.prepare_export_data("PrintExporter")
                    success &= exporter.export(export_data)

                elif isinstance(exporter, ClickHouseExporter):
                    # For ClickHouse, pass tuple with table info
                    export_data = self.prepare_export_data("ClickHouseExporter")
                    if export_data:
                        success &= exporter.export(
                            export_data,
                            table=self.get_clickhouse_table(),
                            column_names=export_data[1],
                            column_type_names=export_data[2]
                        )

            except Exception as e:
                self.log.error(f"Error exporting IOCs to {exporter.__class__.__name__}: {e}")
                success = False

        return success