Nathan Bos

31 papers A* 5A 2B 2Journal 5Unranked 16
YearRankTypeTitle / Venue / Authors
2021 J jnl
CoRR
Nicholas Kantack, Nina Cohen, Nathan Bos, Corey Lowman, James Everett, Timothy Endres
2021 J jnl
Comput. Secur.
Shannon Wasko, Rebecca E. Rhodes, Megan Goforth, Nathan Bos, Hannah P. Cowley, Gerald Matthews, Alice Leung, Satish Iyengar, Jonathon Kopecky
2017 B conf
CogSci
Rebecca E. Rhodes, Isaiah Harbison, Nathan Bos, Celeste Lyn Paul, Clay Fink, Anthony Johnson
2017 J jnl
Games Cult.
Rebecca E. Rhodes, Jonathon Kopecky, Nathan Bos, Jennifer A. McKneely, Abigail S. Gertner, Franklin Zaromb, Alexander Perrone, Jason Spitaletta
2014 A conf
ICWSM
Aurora C. Schmidt, Clay Fink, Nathan Bos
2013 ed.
SBP
Ariel M. Greenberg, William G. Kennedy, Nathan Bos
2013 conf
SocialCom
Clayton Fink, Nathan Bos, Alexander Perrone, Edwina Liu, Jonathon Kopecky
2012 A* conf
CHI
Amy Voida, Nathan Bos, Judith S. Olson, Gary M. Olson, Lauren Dunning
2012 conf
SBP
Barton L. Paulhamus, Alison Ebaugh, C. C. Boylls, Nathan Bos, Sandy Hider, Stephen Giguere
2012 conf
SBP
Clayton Fink, Jonathon Kopecky, Nathan Bos, Max Thomas
2010 B conf
GROUP
Nathan Bos, Ayse G. Buyuktur, Judith S. Olson, Gary M. Olson, Amy Voida
2009 conf
CHI Extended Abstracts
Nathan Bos, Karrie Karahalios, Marcela Musgrove-Chávez, Erika Shehan Poole, John Charles Thomas, Sarita Yardi
2009 J jnl
J. Inf. Technol. Res.
Nathan Bos, Judith S. Olson, Ning Nan, Arik Cheshin
2007 J jnl
J. Comput. Mediat. Commun.
Nathan Bos, Ann Zimmerman, Judith S. Olson, Jude Yew, Jason Yerkie, Erik Dahl, Gary M. Olson
2006 A* conf
CHI
Nathan Bos, Judith S. Olson, Ning Nan, N. Sadat Shami, Susannah Hoch, Erik W. Johnston
2005 conf
CHI Extended Abstracts
Ning Nan, Erik W. Johnston, Judith S. Olson, Nathan Bos
2005 conf
CHI Extended Abstracts
Elaine M. Raybourn, Nathan Bos
2005 conf
AMCIS
Ning Nan, Nathan Bos, Yong-Suk Kim, Arik Cheshin, Judith S. Olson
2005 conf
CHI Extended Abstracts
Nathan Bos, Judith S. Olson, Arik Cheshin, Yong-Suk Kim, Ning Nan, N. Sadat Shami
2004 conf
CASCON
N. Sadat Shami, Nathan Bos, Zach Wright, Susannah Hoch, Kam Yung Kuan, Judith S. Olson, Gary M. Olson
2004 A conf
CSCW
Nathan Bos, N. Sadat Shami, Judith S. Olson, Arik Cheshin, Ning Nan
2003 conf
Hypertext
Harris Wu, Michael D. Gordon, Kurt DeMaagd, Nathan Bos
2002 conf
CSCL
Yael Kali, Nathan Bos, Marcia C. Linn, Jody S. Underwood, Jim Hewitt
2002 A* conf
CHI
Nathan Bos, Judith S. Olson, Darren Gergle, Gary M. Olson, Zach Wright
2002 conf
CSCL
Ryoko Yamaguchi, Nathan Bos, Judy Olson
2002 A* conf
CHI
Jun Zheng, Elizabeth S. Veinott, Nathan Bos, Judith S. Olson, Gary M. Olson
2001 conf
CHI Extended Abstracts
Nathan Bos, Darren Gergle, Judith S. Olson, Gary M. Olson
2001 conf
CHI Extended Abstracts
Jun Zheng, Nathan Bos, Judith S. Olson, Gary M. Olson
1998 A* conf
CHI
Raven Wallace, Elliot Soloway, Joseph Krajcik, Nathan Bos, Joseph Hoffman, Heather Eccleston Hunter, Dan Kiskis, Elisabeth Klann, Greg Peters, David Richardson, Ofer Ronen
1997 conf
CSCL
Jeff Kupperman, Raven Wallace, Nathan Bos
1996 conf
ICLS
Gene Alloway, Nathan Bos, Kathleen Hamel, Tracy Hammerman, Elisabeth Klann, Joseph Krajcik, David Lyons, Terry Madden, Jon Margerum-Leys, James Reed, Nancy Scala, Elliot Soloway, Ioanna Vekiri, Raven Wallace
redb/extractors/js_extractors/js_strings.py
← Index redb/extractors/js_extractors/js_strings.py python
import base64
import bisect
import inspect
import re
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.js_extractor import JSExtractor
from redb.extractors.js_extractors.js_patterns import STRING_PATTERNS, line_offsets

# Local aliases for the compiled patterns this extractor uses. Defined and
# compiled exactly once in js_patterns.STRING_PATTERNS.
_HEX_STRING_RE = STRING_PATTERNS["hex_escape_seq"]
_UNICODE_STRING_RE = STRING_PATTERNS["unicode_escape_seq"]
_CHARCODE_RE = STRING_PATTERNS["charcode_call"]
_BASE64_STRING_RE = STRING_PATTERNS["base64_quoted"]
_CONCAT_STRING_RE = STRING_PATTERNS["concat_chain"]

# Tokeniser used inside _reconstruct_concat to pull each quoted part out of a
# matched concat chain. Compiled once at module load (was recompiled on every
# concat match before).
_CONCAT_TOKEN_RE = re.compile(r'["\']([^"\']*)["\']')


class JSStringsExtractor(JSExtractor):

    def __init__(
        self, filepath, log, exporters=None, index_prefix=None,
        known_benign=False, known_malicious=False, source=None, context=None,
    ):
        super().__init__(
            filepath, log, exporters, index_prefix,
            known_benign, known_malicious, source, context=context,
        )
        self.string_findings = None
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.JS_STRINGS.value

    def _decode_hex_string(self, hex_str):
        """Decode \\x41\\x42 style hex strings."""
        try:
            # Remove \\x prefix and decode
            clean = hex_str.replace('\\x', '')
            return bytes.fromhex(clean).decode('utf-8', errors='replace')
        except Exception:
            return None

    def _decode_unicode_string(self, uni_str):
        """Decode \\u0041\\u0042 style unicode strings."""
        try:
            return uni_str.encode('utf-8').decode('unicode_escape')
        except Exception:
            return None

    def _decode_charcode(self, charcode_str):
        """Decode String.fromCharCode(72, 101, 108, ...) sequences."""
        try:
            codes = [int(c.strip()) for c in charcode_str.split(',') if c.strip().isdigit()]
            return ''.join(chr(c) for c in codes if 0 <= c <= 0x10FFFF)
        except Exception:
            return None

    def _decode_base64(self, b64_str):
        """Attempt to decode base64 string."""
        try:
            decoded = base64.b64decode(b64_str)
            # Check if result is printable text
            text = decoded.decode('utf-8', errors='strict')
            # Only return if it looks like text (>80% printable)
            printable = sum(1 for c in text if c.isprintable() or c in '\n\r\t')
            if printable / len(text) > 0.8:
                return text
        except Exception:
            pass
        return None

    def _reconstruct_concat(self, concat_match):
        """Reconstruct concatenated string parts."""
        try:
            parts = _CONCAT_TOKEN_RE.findall(concat_match)
            return ''.join(parts)
        except Exception:
            return None

    def _find_line_number(self, match_start):
        """1-indexed line number for `match_start`, looked up in O(log L) via
        bisect over `self._line_offsets` (built once per extract() call).

        Replaces the historical `self.js_source[:match_start].count('\\n') + 1`
        which was O(N) per call and quadratic across all matches in a sample.
        """
        return bisect.bisect_right(self._line_offsets, match_start)

    def _scan_text(self, text):
        """Run every encoded-string pattern over `text` and return a list of
        finding dicts. Stateless apart from the per-call `_line_offsets` cache,
        which `_find_line_number` reads — callers must reset it before invoking
        this so line numbers reference the text being scanned, not the previous
        one.
        """
        findings = []

        # Hex-encoded strings
        for m in _HEX_STRING_RE.finditer(text):
            raw = m.group()
            decoded = self._decode_hex_string(raw)
            if decoded and len(decoded) >= 4:
                findings.append({
                    'string': decoded[:4000],
                    'string_raw': raw[:4000],
                    'string_encoding': 'hex',
                    'string_offset': self._find_line_number(m.start()),
                    'string_length': len(decoded),
                    'string_raw_length': len(raw),
                    'string_entropy': self._calculate_text_entropy(decoded),
                })

        # Unicode-encoded strings
        for m in _UNICODE_STRING_RE.finditer(text):
            raw = m.group()
            decoded = self._decode_unicode_string(raw)
            if decoded and len(decoded) >= 3:
                findings.append({
                    'string': decoded[:4000],
                    'string_raw': raw[:4000],
                    'string_encoding': 'unicode',
                    'string_offset': self._find_line_number(m.start()),
                    'string_length': len(decoded),
                    'string_raw_length': len(raw),
                    'string_entropy': self._calculate_text_entropy(decoded),
                })

        # String.fromCharCode sequences
        for m in _CHARCODE_RE.finditer(text):
            raw = m.group()
            decoded = self._decode_charcode(m.group(1))
            if decoded and len(decoded) >= 4:
                findings.append({
                    'string': decoded[:4000],
                    'string_raw': raw[:4000],
                    'string_encoding': 'charcode',
                    'string_offset': self._find_line_number(m.start()),
                    'string_length': len(decoded),
                    'string_raw_length': len(raw),
                    'string_entropy': self._calculate_text_entropy(decoded),
                })

        # Base64-encoded strings
        for m in _BASE64_STRING_RE.finditer(text):
            raw = m.group(0)
            b64_val = m.group(1)
            decoded = self._decode_base64(b64_val)
            if decoded and len(decoded) >= 10:
                findings.append({
                    'string': decoded[:4000],
                    'string_raw': raw[:4000],
                    'string_encoding': 'base64',
                    'string_offset': self._find_line_number(m.start()),
                    'string_length': len(decoded),
                    'string_raw_length': len(raw),
                    'string_entropy': self._calculate_text_entropy(decoded),
                })

        # Concatenated strings (reassembled)
        for m in _CONCAT_STRING_RE.finditer(text):
            raw = m.group()
            reconstructed = self._reconstruct_concat(raw)
            if reconstructed and len(reconstructed) >= 20:
                findings.append({
                    'string': reconstructed[:4000],
                    'string_raw': raw[:4000],
                    'string_encoding': 'concat',
                    'string_offset': self._find_line_number(m.start()),
                    'string_length': len(reconstructed),
                    'string_raw_length': len(raw),
                    'string_entropy': self._calculate_text_entropy(reconstructed),
                })

        return findings

    def extract(self):
        src = self.js_source
        if not src:
            return None

        # Pass 1: raw source. _line_offsets is keyed off whichever text is
        # currently being scanned so _find_line_number resolves to that text.
        self._line_offsets = line_offsets(src)
        findings = self._scan_text(src)

        # Pass 2: deobfuscated text, when the deobfuscator produced something
        # meaningfully different. Same patterns, but a different surface — for
        # samples where the encoded payload is hidden behind an outer wrapper
        # (e.g. array.join() + eval in Vjw0rm/WSH-RAT) only this pass yields
        # any rows at all.
        deobf_text, _ = self._context.deobfuscated
        if deobf_text and deobf_text != src:
            self._line_offsets = line_offsets(deobf_text)
            findings.extend(self._scan_text(deobf_text))

        if not findings:
            return None

        # Deduplicate by decoded string value (raw pass wins on collision: it
        # comes first in `findings`). A string that surfaces only in the
        # deobfuscated text still gets persisted, which is the whole point of
        # the second pass.
        seen_values = set()
        deduped = []
        for f in findings:
            val_key = f['string'][:100]
            if val_key not in seen_values:
                seen_values.add(val_key)
                deduped.append(f)

        self.string_findings = deduped[:500]  # Limit per file
        # Publish to the shared context so post-loop consumers (notably the IOC
        # plumbing in workers.py) can scrape the decoded strings without
        # holding a reference to this extractor instance.
        self._context.decoded_strings = self.string_findings
        return self.string_findings

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type == "ClickHouseExporter":
            if not self.string_findings:
                return None

            data = []
            for f in self.string_findings:
                data.append([
                    self.sha256,
                    f['string'],
                    f['string_raw'],
                    f['string_encoding'],
                    f['string_offset'],
                    f['string_length'],
                    f['string_raw_length'],
                    f['string_entropy'],
                ])

            column_names = [
                "sha256",
                "string",
                "string_raw",
                "string_encoding",
                "string_offset",
                "string_length",
                "string_raw_length",
                "string_entropy",
            ]

            column_type_names = [
                "FixedString(64)",
                "String",
                "String",
                "LowCardinality(String)",
                "UInt64",
                "UInt32",
                "UInt32",
                "Float32",
            ]

            return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "code_binja_strings_raw"