Naomi Inoue

32 papers A* 6Misc 2Journal 5Unranked 19
YearRankTypeTitle / Venue / Authors
2014 Misc conf
ICASSP
Mehrdad Panahpour Tehrani, Akio Ishikawa, Makoto Okui, Naomi Inoue, Keita Takahashi, Toshiaki Fujii
2012 conf
SD&A
Masahiro Kawakita, Shoichiro Iwasawa, M. Sakai, Yasuyuki Haino, Masahito Sato, Naomi Inoue
2012 conf
3DTV-Conference
Mehrdad Panahpour Tehrani, Akio Ishikawa, Masahiro Kawakita, Naomi Inoue, Toshiaki Fujii
2012 J jnl
Pattern Recognit.
Sabri Gurbuz, Erhan Öztop, Naomi Inoue
2010 conf
VRCAI
Naomi Inoue
2009 J jnl
IEICE Trans. Inf. Syst.
Naomi Inoue
2009 Misc conf
ICASSP
Ryouichi Nishimura, Hiroaki Kato, Naomi Inoue
2009 conf
SIGGRAPH Emerging Technologies
Roberto Lopez-Gulliver, Shunsuke Yoshida, Sumio Yano, Naomi Inoue
2008 conf
3DUI
Roberto Lopez-Gulliver, Shunsuke Yoshida, Sumio Yano, Naomi Inoue
2008 conf
SIGGRAPH Posters
Roberto Lopez-Gulliver, Shunsuke Yoshida, Sumio Yano, Naomi Inoue
2007 conf
ICASSP (2)
Sabri Gurbuz, Naomi Inoue
2007 conf
HCI (9)
Noriko Suzuki, Ichiro Umata, Tatsuya Kitamura, Hiroshi Ando, Naomi Inoue
2006 conf
AUIC
Rodney Berry, Masahide Naemura, Yuichi Kobayashi, Masahiro Tada, Naomi Inoue, Yusuf Pisan, Ernest A. Edmonds
2006 J jnl
Multim. Syst.
Rodney Berry, Mao Makino, Naoto Hikawa, Masami Suzuki, Naomi Inoue
2003 conf
SAINT Workshops
Kazunori Matsumoto, Shigeki Muramatsu, Naomi Inoue, Hideki Mori
2003 conf
SIGGRAPH Web Graphics
Arei Kobayashi, Satoru Takagi, Naomi Inoue
2003 A* conf
ACM Multimedia
Keiichiro Hoashi, Kazunori Matsumoto, Naomi Inoue
2002 A* conf
SIGIR
Keiichiro Hoashi, Erik Zeitler, Naomi Inoue
2001 J jnl
Comput. Humanit.
Masami Suzuki, Naomi Inoue, Kazuo Hashimoto
2001 A* conf
SIGIR
Keiichiro Hoashi, Kazunori Matsumoto, Naomi Inoue, Kazuo Hashimoto
2000 A* conf
SIGIR
Keiichiro Hoashi, Kazunori Matsumoto, Naomi Inoue, Kazuo Hashimoto
2000 conf
TREC
Keiichiro Hoashi, Kazunori Matsumoto, Naomi Inoue, Kazuo Hashimoto, Takashi Hasegawa, Katsuhiko Shirai
1999 conf
EUROSPEECH
Fumihiro Yato, Naomi Inoue, Kazuo Hashimoto
1999 conf
TREC
Keiichiro Hoashi, Kazunori Matsumoto, Naomi Inoue, Kazuo Hashimoto
1999 A* conf
SIGIR
Keiichiro Hoashi, Kazunori Matsumoto, Naomi Inoue, Kazuo Hashimoto
1998 conf
TREC
Keiichiro Hoashi, Kazunori Matsumoto, Naomi Inoue, Kazuo Hashimoto
1995 conf
EUROSPEECH
Masami Suzuki, Naomi Inoue, Fumihiro Yato, Kazuya Takeda, Seiichi Yamamoto
1995 J jnl
IEICE Trans. Inf. Syst.
Shingo Kuroiwa, Kazuya Takeda, Masaki Naito, Naomi Inoue, Seiichi Yamamoto
1993 conf
EUROSPEECH
Shingo Kuroiwa, Kazuya Takeda, Naomi Inoue, Izuru Nogaito, Seiichi Yamamoto, Makoto Shozakai, Kunihiko Owa, Masahiko Takahashi, Ryuuji Matsumoto
1993 conf
EUROSPEECH
Kazuya Takeda, Naomi Inoue, Shingo Kuroiwa, Tomohiro Konuma, Seiichi Yamamoto
1991 A* conf
ACL
Naomi Inoue
1989 conf
EUROSPEECH
Naomi Inoue, Tsuyoshi Morimoto, Kentaro Ogura
redb/extractors/js_extractors/js_strings.py
← Index redb/extractors/js_extractors/js_strings.py python
import base64
import bisect
import inspect
import re
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor
from redb.extractors.js_extractors.js_patterns import STRING_PATTERNS, line_offsets

# Local aliases for the compiled patterns this extractor uses. Defined and
# compiled exactly once in js_patterns.STRING_PATTERNS.
_HEX_STRING_RE = STRING_PATTERNS["hex_escape_seq"]
_UNICODE_STRING_RE = STRING_PATTERNS["unicode_escape_seq"]
_CHARCODE_RE = STRING_PATTERNS["charcode_call"]
_BASE64_STRING_RE = STRING_PATTERNS["base64_quoted"]
_CONCAT_STRING_RE = STRING_PATTERNS["concat_chain"]

# Tokeniser used inside _reconstruct_concat to pull each quoted part out of a
# matched concat chain. Compiled once at module load (was recompiled on every
# concat match before).
_CONCAT_TOKEN_RE = re.compile(r'["\']([^"\']*)["\']')


class JSStringsExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.string_findings = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_STRINGS.value

    def _decode_hex_string(self, hex_str):
        """Decode \\x41\\x42 style hex strings."""
        try:
            # Remove \\x prefix and decode
            clean = hex_str.replace('\\x', '')
            return bytes.fromhex(clean).decode('utf-8', errors='replace')
        except Exception:
            return None

    def _decode_unicode_string(self, uni_str):
        """Decode \\u0041\\u0042 style unicode strings."""
        try:
            return uni_str.encode('utf-8').decode('unicode_escape')
        except Exception:
            return None

    def _decode_charcode(self, charcode_str):
        """Decode String.fromCharCode(72, 101, 108, ...) sequences."""
        try:
            codes = [int(c.strip()) for c in charcode_str.split(',') if c.strip().isdigit()]
            return ''.join(chr(c) for c in codes if 0 <= c <= 0x10FFFF)
        except Exception:
            return None

    def _decode_base64(self, b64_str):
        """Attempt to decode base64 string."""
        try:
            decoded = base64.b64decode(b64_str)
            # Check if result is printable text
            text = decoded.decode('utf-8', errors='strict')
            # Only return if it looks like text (>80% printable)
            printable = sum(1 for c in text if c.isprintable() or c in '\n\r\t')
            if printable / len(text) > 0.8:
                return text
        except Exception:
            pass
        return None

    def _reconstruct_concat(self, concat_match):
        """Reconstruct concatenated string parts."""
        try:
            parts = _CONCAT_TOKEN_RE.findall(concat_match)
            return ''.join(parts)
        except Exception:
            return None

    def _find_line_number(self, match_start):
        """1-indexed line number for `match_start`, looked up in O(log L) via
        bisect over `self._line_offsets` (built once per extract() call).

        Replaces the historical `self.js_source[:match_start].count('\\n') + 1`
        which was O(N) per call and quadratic across all matches in a sample.
        """
        return bisect.bisect_right(self._line_offsets, match_start)

    def _scan_text(self, text):
        """Run every encoded-string pattern over `text` and return a list of
        finding dicts. Stateless apart from the per-call `_line_offsets` cache,
        which `_find_line_number` reads — callers must reset it before invoking
        this so line numbers reference the text being scanned, not the previous
        one.
        """
        findings = []

        # Hex-encoded strings
        for m in _HEX_STRING_RE.finditer(text):
            raw = m.group()
            decoded = self._decode_hex_string(raw)
            if decoded and len(decoded) >= 4:
                findings.append({
                    'string': decoded[:4000],
                    'string_raw': raw[:4000],
                    'string_encoding': 'hex',
                    'string_offset': self._find_line_number(m.start()),
                    'string_length': len(decoded),
                    'string_raw_length': len(raw),
                    'string_entropy': self._calculate_text_entropy(decoded),
                })

        # Unicode-encoded strings
        for m in _UNICODE_STRING_RE.finditer(text):
            raw = m.group()
            decoded = self._decode_unicode_string(raw)
            if decoded and len(decoded) >= 3:
                findings.append({
                    'string': decoded[:4000],
                    'string_raw': raw[:4000],
                    'string_encoding': 'unicode',
                    'string_offset': self._find_line_number(m.start()),
                    'string_length': len(decoded),
                    'string_raw_length': len(raw),
                    'string_entropy': self._calculate_text_entropy(decoded),
                })

        # String.fromCharCode sequences
        for m in _CHARCODE_RE.finditer(text):
            raw = m.group()
            decoded = self._decode_charcode(m.group(1))
            if decoded and len(decoded) >= 4:
                findings.append({
                    'string': decoded[:4000],
                    'string_raw': raw[:4000],
                    'string_encoding': 'charcode',
                    'string_offset': self._find_line_number(m.start()),
                    'string_length': len(decoded),
                    'string_raw_length': len(raw),
                    'string_entropy': self._calculate_text_entropy(decoded),
                })

        # Base64-encoded strings
        for m in _BASE64_STRING_RE.finditer(text):
            raw = m.group(0)
            b64_val = m.group(1)
            decoded = self._decode_base64(b64_val)
            if decoded and len(decoded) >= 10:
                findings.append({
                    'string': decoded[:4000],
                    'string_raw': raw[:4000],
                    'string_encoding': 'base64',
                    'string_offset': self._find_line_number(m.start()),
                    'string_length': len(decoded),
                    'string_raw_length': len(raw),
                    'string_entropy': self._calculate_text_entropy(decoded),
                })

        # Concatenated strings (reassembled)
        for m in _CONCAT_STRING_RE.finditer(text):
            raw = m.group()
            reconstructed = self._reconstruct_concat(raw)
            if reconstructed and len(reconstructed) >= 20:
                findings.append({
                    'string': reconstructed[:4000],
                    'string_raw': raw[:4000],
                    'string_encoding': 'concat',
                    'string_offset': self._find_line_number(m.start()),
                    'string_length': len(reconstructed),
                    'string_raw_length': len(raw),
                    'string_entropy': self._calculate_text_entropy(reconstructed),
                })

        return findings

    def extract(self):
        src = self.js_source
        if not src:
            return None

        # Pass 1: raw source. _line_offsets is keyed off whichever text is
        # currently being scanned so _find_line_number resolves to that text.
        self._line_offsets = line_offsets(src)
        findings = self._scan_text(src)

        # Pass 2: deobfuscated text, when the deobfuscator produced something
        # meaningfully different. Same patterns, but a different surface — for
        # samples where the encoded payload is hidden behind an outer wrapper
        # (e.g. array.join() + eval in Vjw0rm/WSH-RAT) only this pass yields
        # any rows at all.
        deobf_text, _ = self._context.deobfuscated
        if deobf_text and deobf_text != src:
            self._line_offsets = line_offsets(deobf_text)
            findings.extend(self._scan_text(deobf_text))

        if not findings:
            return None

        # Deduplicate by decoded string value (raw pass wins on collision: it
        # comes first in `findings`). A string that surfaces only in the
        # deobfuscated text still gets persisted, which is the whole point of
        # the second pass.
        seen_values = set()
        deduped = []
        for f in findings:
            val_key = f['string'][:100]
            if val_key not in seen_values:
                seen_values.add(val_key)
                deduped.append(f)

        self.string_findings = deduped[:500]  # Limit per file
        # Publish to the shared context so post-loop consumers (notably the IOC
        # plumbing in workers.py) can scrape the decoded strings without
        # holding a reference to this extractor instance.
        self._context.decoded_strings = self.string_findings
        return self.string_findings

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type == "ClickHouseExporter":
            if not self.string_findings:
                return None

            data = []
            for f in self.string_findings:
                data.append([
                    self.sha256,
                    f['string'],
                    f['string_raw'],
                    f['string_encoding'],
                    f['string_offset'],
                    f['string_length'],
                    f['string_raw_length'],
                    f['string_entropy'],
                ])

            column_names = [
                "sha256",
                "string",
                "string_raw",
                "string_encoding",
                "string_offset",
                "string_length",
                "string_raw_length",
                "string_entropy",
            ]

            column_type_names = [
                "FixedString(64)",
                "String",
                "String",
                "LowCardinality(String)",
                "UInt64",
                "UInt32",
                "UInt32",
                "Float32",
            ]

            return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_binja_strings_raw"