Jaehoon Cha

23 papers A* 1Misc 1Journal 15Unranked 6
YearRankTypeTitle / Venue / Authors
2026 conf
ISSCC
Hyunsu Park, Sungkwon Lee, Kyunghoon Kim, Jin-Youp Cha, Hyeongjun Ko, Jaehyeok Yang, Seongjin Kim, Youngtaek Kim, Minsoo Park, Gangsik Lee, Keonho Lee, Sanghoon Lee, Jiseok Lee, Gyunam Jeon, Sera Jeong, Yongsuk Joo, Jaehoon Cha, Seonwoo Hwang, Seulgi Kim, Eunji Song, Junghwan Ji, Byungjun Kang, Boram Kim, Sang-Yeon Byeon, Hankyu Chi, Hyungsoo Kim, Jonghwan Kim
2026 J jnl
Expert Syst. Appl.
Jongyun Byun, Jaehoon Cha, Jeyan Thiyagalingam, Hyeon-Joon Kim, Changhyun Jun
2026 conf
EACL (Findings)
Kuangdai Leng, Jia Bi, Samuel Pinilla, Jaehoon Cha
2025 J jnl
Nat. Mac. Intell.
Jaehoon Cha, Jinhae Park, Samuel Pinilla, Kyle L. Morris, Christopher S. Allen, Mark I. Wilkinson, Jeyan Thiyagalingam
2025 Misc conf
ICASSP
Jaehoon Cha, Siu Lun Yeung, Siddharth Dhanpal, Jeyan Thiyagalingam
2024 conf
ISSCC
Jaehyeok Yang, Hyeongjun Ko, Kyunghoon Kim, Hyunsu Park, Jihwan Park, Ji-Hyo Kang, Jin-Youp Cha, Seongjin Kim, Youngtaek Kim, Minsoo Park, Gangsik Lee, Keonho Lee, Sanghoon Lee, Gyunam Jeon, Sera Jeong, Yongsuk Joo, Jaehoon Cha, Seonwoo Hwang, Boram Kim, Sang-Yeon Byeon, Sungkwon Lee, Hyeonyeol Park, Joohwan Cho, Jonghwan Kim
2024 J jnl
CoRR
Kuangdai Leng, Jia Bi, Jaehoon Cha, Samuel Pinilla, Jeyan Thiyagalingam
2024 J jnl
Symmetry
Kuangdai Leng, Jia Bi, Jaehoon Cha, Samuel Pinilla, Jeyan Thiyagalingam
2024 J jnl
CoRR
Saiful Khan, Scott Jones, Benjamin Bach, Jaehoon Cha, Min Chen, Julie Meikle, Jonathan C. Roberts, Jeyan Thiyagalingam, Jo Wood, Panagiotis D. Ritsos
2023 J jnl
IEEE Trans. Geosci. Remote. Sens.
Jongyun Byun, Changhyun Jun, Jinwon Kim, Jaehoon Cha, Roya Narimani
2023 A* conf
ICML
Jaehoon Cha, Jeyan Thiyagalingam
2022 J jnl
IEEE J. Solid State Circuits
Jihyo Kang, Jaehyeok Yang, Kyunghoon Kim, Joo-Hyung Chae, Gang-Sik Lee, Sang-Yeon Byeon, Boram Kim, Dong-Hyun Kim, Youngtaek Kim, Yeongmuk Cho, Junghwan Ji, Sera Jeong, Jaehoon Cha, Minsoo Park, Hongdeuk Kim, Sijun Park, Sunho Kim, Hae-Kang Jung, Jieun Jang, Sangkwon Lee, Hyungsoo Kim, Joo-Hwan Cho, Junhyun Chun, Seon-Yong Cha
2022 J jnl
Appl. Soft Comput.
Jaehoon Cha, Enggee Lim
2022 J jnl
CoRR
Jaehoon Cha, Jeyan Thiyagalingam
2021 conf
ISSCC
Kyunghoon Kim, Joo-Hyung Chae, Jaehyeok Yang, Jihyo Kang, Gang-Sik Lee, Sang-Yeon Byeon, Youngtaek Kim, Boram Kim, Dong-Hyun Kim, Yeongmuk Cho, Kangmoo Choi, Hyeongyeol Park, Junghwan Ji, Sera Jeong, Yongsuk Joo, Jaehoon Cha, Minsoo Park, Hongdeuk Kim, Sijun Park, Kyubong Kong, Sunho Kim, Sangkwon Lee, Junhyun Chun, Hyungsoo Kim, Seon-Yong Cha
2020 J jnl
Mach. Learn. Sci. Technol.
Jaehoon Cha, Kyeong Soo Kim, Sanghyuk Lee
2019 conf
ISOCC
Sanghyuk Lee, Jaehoon Cha, Kyeong Soo Kim
2019 J jnl
CoRR
Jaehoon Cha, Kyeong Soo Kim, Sanghyuk Lee
2019 J jnl
CoRR
Jaehoon Cha, Kyeong Soo Kim, Sanghyuk Lee
2017 J jnl
Symmetry
Sanghyuk Lee, Jaehoon Cha, Nipon Theera-Umpon, Kyeong Soo Kim
2017 J jnl
CoRR
Kyeong Soo Kim, Ruihao Wang, Zhenghang Zhong, Zikun Tan, Haowei Song, Jaehoon Cha, Sanghyuk Lee
2017 conf
IFSA-SCIS
Jaehoon Cha, Sanghyuk Lee, Kyeong Soo Kim, Witold Pedrycz
2007 J jnl
IEEE J. Solid State Circuits
Ki-Won Lee, Joo-Hwan Cho, Byoung-Jin Choi, Geun-Il Lee, Ho-Don Jung, Woo-Young Lee, Ki-Chon Park, Yongsuk Joo, Jaehoon Cha, Young-Jung Choi, Patrick B. Moran, Jin-Hong Ahn
redb/extractors/js_extractors/js_strings.py
← Index redb/extractors/js_extractors/js_strings.py python
import base64
import bisect
import inspect
import re
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor
from redb.extractors.js_extractors.js_patterns import STRING_PATTERNS, line_offsets

# Local aliases for the compiled patterns this extractor uses. Defined and
# compiled exactly once in js_patterns.STRING_PATTERNS.
_HEX_STRING_RE = STRING_PATTERNS["hex_escape_seq"]
_UNICODE_STRING_RE = STRING_PATTERNS["unicode_escape_seq"]
_CHARCODE_RE = STRING_PATTERNS["charcode_call"]
_BASE64_STRING_RE = STRING_PATTERNS["base64_quoted"]
_CONCAT_STRING_RE = STRING_PATTERNS["concat_chain"]

# Tokeniser used inside _reconstruct_concat to pull each quoted part out of a
# matched concat chain. Compiled once at module load (was recompiled on every
# concat match before).
_CONCAT_TOKEN_RE = re.compile(r'["\']([^"\']*)["\']')


class JSStringsExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.string_findings = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_STRINGS.value

    def _decode_hex_string(self, hex_str):
        """Decode \\x41\\x42 style hex strings."""
        try:
            # Remove \\x prefix and decode
            clean = hex_str.replace('\\x', '')
            return bytes.fromhex(clean).decode('utf-8', errors='replace')
        except Exception:
            return None

    def _decode_unicode_string(self, uni_str):
        """Decode \\u0041\\u0042 style unicode strings."""
        try:
            return uni_str.encode('utf-8').decode('unicode_escape')
        except Exception:
            return None

    def _decode_charcode(self, charcode_str):
        """Decode String.fromCharCode(72, 101, 108, ...) sequences."""
        try:
            codes = [int(c.strip()) for c in charcode_str.split(',') if c.strip().isdigit()]
            return ''.join(chr(c) for c in codes if 0 <= c <= 0x10FFFF)
        except Exception:
            return None

    def _decode_base64(self, b64_str):
        """Attempt to decode base64 string."""
        try:
            decoded = base64.b64decode(b64_str)
            # Check if result is printable text
            text = decoded.decode('utf-8', errors='strict')
            # Only return if it looks like text (>80% printable)
            printable = sum(1 for c in text if c.isprintable() or c in '\n\r\t')
            if printable / len(text) > 0.8:
                return text
        except Exception:
            pass
        return None

    def _reconstruct_concat(self, concat_match):
        """Reconstruct concatenated string parts."""
        try:
            parts = _CONCAT_TOKEN_RE.findall(concat_match)
            return ''.join(parts)
        except Exception:
            return None

    def _find_line_number(self, match_start):
        """1-indexed line number for `match_start`, looked up in O(log L) via
        bisect over `self._line_offsets` (built once per extract() call).

        Replaces the historical `self.js_source[:match_start].count('\\n') + 1`
        which was O(N) per call and quadratic across all matches in a sample.
        """
        return bisect.bisect_right(self._line_offsets, match_start)

    def _scan_text(self, text):
        """Run every encoded-string pattern over `text` and return a list of
        finding dicts. Stateless apart from the per-call `_line_offsets` cache,
        which `_find_line_number` reads — callers must reset it before invoking
        this so line numbers reference the text being scanned, not the previous
        one.
        """
        findings = []

        # Hex-encoded strings
        for m in _HEX_STRING_RE.finditer(text):
            raw = m.group()
            decoded = self._decode_hex_string(raw)
            if decoded and len(decoded) >= 4:
                findings.append({
                    'string': decoded[:4000],
                    'string_raw': raw[:4000],
                    'string_encoding': 'hex',
                    'string_offset': self._find_line_number(m.start()),
                    'string_length': len(decoded),
                    'string_raw_length': len(raw),
                    'string_entropy': self._calculate_text_entropy(decoded),
                })

        # Unicode-encoded strings
        for m in _UNICODE_STRING_RE.finditer(text):
            raw = m.group()
            decoded = self._decode_unicode_string(raw)
            if decoded and len(decoded) >= 3:
                findings.append({
                    'string': decoded[:4000],
                    'string_raw': raw[:4000],
                    'string_encoding': 'unicode',
                    'string_offset': self._find_line_number(m.start()),
                    'string_length': len(decoded),
                    'string_raw_length': len(raw),
                    'string_entropy': self._calculate_text_entropy(decoded),
                })

        # String.fromCharCode sequences
        for m in _CHARCODE_RE.finditer(text):
            raw = m.group()
            decoded = self._decode_charcode(m.group(1))
            if decoded and len(decoded) >= 4:
                findings.append({
                    'string': decoded[:4000],
                    'string_raw': raw[:4000],
                    'string_encoding': 'charcode',
                    'string_offset': self._find_line_number(m.start()),
                    'string_length': len(decoded),
                    'string_raw_length': len(raw),
                    'string_entropy': self._calculate_text_entropy(decoded),
                })

        # Base64-encoded strings
        for m in _BASE64_STRING_RE.finditer(text):
            raw = m.group(0)
            b64_val = m.group(1)
            decoded = self._decode_base64(b64_val)
            if decoded and len(decoded) >= 10:
                findings.append({
                    'string': decoded[:4000],
                    'string_raw': raw[:4000],
                    'string_encoding': 'base64',
                    'string_offset': self._find_line_number(m.start()),
                    'string_length': len(decoded),
                    'string_raw_length': len(raw),
                    'string_entropy': self._calculate_text_entropy(decoded),
                })

        # Concatenated strings (reassembled)
        for m in _CONCAT_STRING_RE.finditer(text):
            raw = m.group()
            reconstructed = self._reconstruct_concat(raw)
            if reconstructed and len(reconstructed) >= 20:
                findings.append({
                    'string': reconstructed[:4000],
                    'string_raw': raw[:4000],
                    'string_encoding': 'concat',
                    'string_offset': self._find_line_number(m.start()),
                    'string_length': len(reconstructed),
                    'string_raw_length': len(raw),
                    'string_entropy': self._calculate_text_entropy(reconstructed),
                })

        return findings

    def extract(self):
        src = self.js_source
        if not src:
            return None

        # Pass 1: raw source. _line_offsets is keyed off whichever text is
        # currently being scanned so _find_line_number resolves to that text.
        self._line_offsets = line_offsets(src)
        findings = self._scan_text(src)

        # Pass 2: deobfuscated text, when the deobfuscator produced something
        # meaningfully different. Same patterns, but a different surface — for
        # samples where the encoded payload is hidden behind an outer wrapper
        # (e.g. array.join() + eval in Vjw0rm/WSH-RAT) only this pass yields
        # any rows at all.
        deobf_text, _ = self._context.deobfuscated
        if deobf_text and deobf_text != src:
            self._line_offsets = line_offsets(deobf_text)
            findings.extend(self._scan_text(deobf_text))

        if not findings:
            return None

        # Deduplicate by decoded string value (raw pass wins on collision: it
        # comes first in `findings`). A string that surfaces only in the
        # deobfuscated text still gets persisted, which is the whole point of
        # the second pass.
        seen_values = set()
        deduped = []
        for f in findings:
            val_key = f['string'][:100]
            if val_key not in seen_values:
                seen_values.add(val_key)
                deduped.append(f)

        self.string_findings = deduped[:500]  # Limit per file
        # Publish to the shared context so post-loop consumers (notably the IOC
        # plumbing in workers.py) can scrape the decoded strings without
        # holding a reference to this extractor instance.
        self._context.decoded_strings = self.string_findings
        return self.string_findings

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type == "ClickHouseExporter":
            if not self.string_findings:
                return None

            data = []
            for f in self.string_findings:
                data.append([
                    self.sha256,
                    f['string'],
                    f['string_raw'],
                    f['string_encoding'],
                    f['string_offset'],
                    f['string_length'],
                    f['string_raw_length'],
                    f['string_entropy'],
                ])

            column_names = [
                "sha256",
                "string",
                "string_raw",
                "string_encoding",
                "string_offset",
                "string_length",
                "string_raw_length",
                "string_entropy",
            ]

            column_type_names = [
                "FixedString(64)",
                "String",
                "String",
                "LowCardinality(String)",
                "UInt64",
                "UInt32",
                "UInt32",
                "Float32",
            ]

            return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_binja_strings_raw"