Radford M. Neal

35 papers A 1B 1Journal 23Unranked 6
YearRankTypeTitle / Venue / Authors
2025 J jnl
J. Appl. Probab.
Radford M. Neal, Jeffrey S. Rosenthal
2020 J jnl
CoRR
Radford M. Neal
2015 J jnl
CoRR
Radford M. Neal
2015 J jnl
CoRR
Radford M. Neal
2014 J jnl
Stat. Comput.
Babak Shahbaba, Shiwei Lan, Wesley O. Johnson, Radford M. Neal
2013 J jnl
CoRR
Michael Harvey, Radford M. Neal
2012 J jnl
CoRR
Chunyi Wang, Radford M. Neal
2011 J jnl
CoRR
Radford M. Neal
2009 J jnl
J. Mach. Learn. Res.
Babak Shahbaba, Radford M. Neal
2007 J jnl
Bioinform.
Jennifer Listgarten, Radford M. Neal, Sam T. Roweis, Peter Wong, Andrew Emili
2007 J jnl
Technometrics
Radford M. Neal
2006 conf
NIPS
Jennifer Listgarten, Radford M. Neal, Sam T. Roweis, Rachel Puckrin, Sean Cutler
2006 J jnl
BMC Bioinform.
Babak Shahbaba, Radford M. Neal
2006 ch.
Feature Extraction
Radford M. Neal, Jianguo Zhang
2006 conf
NIPS
Edward Meeds, Zoubin Ghahramani, Radford M. Neal, Sam T. Roweis
2005 conf
MLCW
Radford M. Neal
2004 conf
NIPS
Jennifer Listgarten, Radford M. Neal, Sam T. Roweis, Andrew Emili
2003 conf
NIPS
Radford M. Neal, Matthew J. Beal, Sam T. Roweis
2001 J jnl
Stat. Comput.
Radford M. Neal
2000 A conf
UAI
Michael Harvey, Radford M. Neal
2000 J jnl
J. Artif. Intell. Res.
Radford M. Neal
1998 ch.
Learning in Graphical Models
Radford M. Neal, Geoffrey E. Hinton
1998 J jnl
ACM Trans. Inf. Syst.
Alistair Moffat, Radford M. Neal, Ian H. Witten
1998 ch.
Learning in Graphical Models
Radford M. Neal
1997 J jnl
Neural Comput.
Radford M. Neal, Peter Dayan
1996 J jnl
Stat. Comput.
Radford M. Neal
1995 B conf
Data Compression Conference
Alistair Moffat, Radford M. Neal, Ian H. Witten
1995
Radford M. Neal
1995 J jnl
Neural Comput.
Peter Dayan, Geoffrey E. Hinton, Radford M. Neal, Richard S. Zemel
1993 J jnl
IEEE Trans. Inf. Theory
Radford M. Neal
1992 J jnl
Neural Comput.
Radford M. Neal
1992 conf
NIPS
Radford M. Neal
1992 J jnl
Artif. Intell.
Radford M. Neal
1987 J jnl
Commun. ACM
Ian H. Witten, Radford M. Neal, John G. Cleary
1983 J jnl
ACM SIGOPS Oper. Syst. Rev.
Ian H. Witten, Graham M. Birtwistle, John G. Cleary, David R. Hill, Danny Levinson, Greg Lomow, Radford M. Neal, Murray Peterson, Brian W. Unger, Brian Wyvill
src/main.rs
← Index src/main.rs rust
mod highlight;

use std::collections::BTreeMap;
use std::fs;
use std::io::Read;
use std::path::{Path, PathBuf};
use std::time::Instant;

use walkdir::WalkDir;

use highlight::highlight_code;

/// Above this threshold the file is read in full and classified
/// (text or binary) based on its complete content.
const FULL_READ_LIMIT: u64 = 4_000_000; // 4 MB
/// For files larger than the threshold above, we only read a preview
/// of this size (so we never load huge files fully into memory).
const PREVIEW_BYTES: usize = 16_384;
/// How many bytes of binary content to show in the hex dump.
const HEXDUMP_BYTES: usize = 4096;

struct FileResult {
    rel_path: PathBuf, // relative path of the original file
    out_rel: PathBuf,  // relative path of the generated .html file (inside out/)
    label: String,     // label shown in the badge (language, or "binary")
}

enum Unreadable {
    IoError(PathBuf),
}

enum Content {
    /// Successfully decoded text. `truncated` indicates whether the file
    /// is larger than what was actually read (very large files: preview only).
    Text { content: String, truncated: bool, total_size: u64 },
    /// Content that isn't valid UTF-8 text: we show a hex dump instead
    /// of trying to decode it.
    Binary { preview: Vec<u8>, total_size: u64, truncated: bool },
}

/// Reads a file safely: for small files, reads everything and classifies
/// text/binary based on the full content; for large files, reads only a
/// prefix, to avoid loading huge files into memory.
fn read_content(path: &Path, total_size: u64) -> std::io::Result<Content> {
    if total_size <= FULL_READ_LIMIT {
        let bytes = fs::read(path)?;
        match String::from_utf8(bytes) {
            Ok(s) => Ok(Content::Text {
                content: s,
                truncated: false,
                total_size,
            }),
            Err(e) => Ok(Content::Binary {
                preview: e.into_bytes(),
                total_size,
                truncated: false,
            }),
        }
    } else {
        let mut f = fs::File::open(path)?;
        let mut buf = vec![0u8; PREVIEW_BYTES];
        let n = f.read(&mut buf)?;
        buf.truncate(n);

        match std::str::from_utf8(&buf) {
            Ok(s) => Ok(Content::Text {
                content: s.to_string(),
                truncated: true,
                total_size,
            }),
            Err(e) => {
                let valid_up_to = e.valid_up_to();
                // If the error is only due to cutting a multi-byte
                // character in half (common when reading an arbitrary
                // prefix of a text file), still treat it as text.
                if valid_up_to > 0 && buf.len() - valid_up_to <= 4 {
                    let s = std::str::from_utf8(&buf[..valid_up_to]).unwrap().to_string();
                    Ok(Content::Text {
                        content: s,
                        truncated: true,
                        total_size,
                    })
                } else {
                    Ok(Content::Binary {
                        preview: buf,
                        total_size,
                        truncated: true,
                    })
                }
            }
        }
    }
}

/// Tree node used to build the index (index.html)
#[derive(Default)]
struct Node {
    dirs: BTreeMap<String, Node>,
    files: Vec<(String, String, PathBuf)>, // (display_name, label, href relative to the index)
}

impl Node {
    fn insert_file(&mut self, components: &[String], display_name: String, label: String, href: PathBuf) {
        if components.is_empty() {
            self.files.push((display_name, label, href));
        } else {
            let head = &components[0];
            let child = self.dirs.entry(head.clone()).or_default();
            child.insert_file(&components[1..], display_name, label, href);
        }
    }

    fn render(&self, out: &mut String) {
        out.push_str("<ul class=\"tree\">\n");
        for (name, node) in &self.dirs {
            out.push_str(&format!(
                "<li class=\"dir\"><span class=\"dir-name\">\u{1F4C1} {}</span>\n",
                escape_html(name)
            ));
            node.render(out);
            out.push_str("</li>\n");
        }
        for (name, label, href) in &self.files {
            out.push_str(&format!(
                "<li class=\"file\"><a href=\"{}\">\u{1F4C4} {}</a> <span class=\"badge\">{}</span></li>\n",
                href.to_string_lossy().replace('\\', "/"),
                escape_html(name),
                escape_html(label)
            ));
        }
        out.push_str("</ul>\n");
    }
}

fn main() {
    let args: Vec<String> = std::env::args().collect();
    let input_dir = args.get(1).cloned().unwrap_or_else(|| ".".to_string());
    let output_dir = args.get(2).cloned().unwrap_or_else(|| "out".to_string());
    // Extra directory names to exclude, optional, comma-separated
    // (e.g. "node_modules,.git"). Nothing is excluded by default:
    // the generator includes everything it finds, including
    // __pycache__, compiled files, binaries, etc.
    let extra_excludes: Vec<String> = args
        .get(3)
        .map(|s| s.split(',').map(|p| p.trim().to_string()).filter(|p| !p.is_empty()).collect())
        .unwrap_or_default();

    let input = PathBuf::from(&input_dir);
    let output = PathBuf::from(&output_dir);

    if !input.is_dir() {
        eprintln!("Error: '{}' is not a valid directory.", input.display());
        std::process::exit(1);
    }

    if let Err(e) = fs::create_dir_all(&output) {
        eprintln!("Error creating the output directory: {e}");
        std::process::exit(1);
    }

    // Absolute path of the output directory, so we can exclude it from the
    // scan in case it's nested inside the input directory (e.g. input=".",
    // output="out"). This is the only automatic exclusion and it can't be
    // turned off: it only prevents the program from processing the files
    // it is currently writing.
    let output_abs = fs::canonicalize(&output).unwrap_or_else(|_| output.clone());

    println!("Scanning '{}' -> '{}'", input.display(), output.display());
    if !extra_excludes.is_empty() {
        println!("Directories excluded on request: {}", extra_excludes.join(", "));
    }
    let start = Instant::now();

    let mut results: Vec<FileResult> = Vec::new();
    let mut unreadable: Vec<Unreadable> = Vec::new();
    let mut n_binary = 0usize;
    let mut n_truncated = 0usize;

    for entry in WalkDir::new(&input)
        .into_iter()
        .filter_entry(|e| {
            if e.file_type().is_dir() {
                if let Some(name) = e.file_name().to_str() {
                    if extra_excludes.iter().any(|x| x == name) {
                        return false;
                    }
                }
                if let Ok(abs) = fs::canonicalize(e.path()) {
                    if abs == output_abs {
                        return false;
                    }
                }
            }
            true
        })
    {
        let entry = match entry {
            Ok(e) => e,
            Err(_) => continue,
        };

        if !entry.file_type().is_file() {
            continue;
        }

        let path = entry.path();
        let rel_path = match path.strip_prefix(&input) {
            Ok(p) => p.to_path_buf(),
            Err(_) => path.to_path_buf(),
        };

        let meta = match fs::metadata(path) {
            Ok(m) => m,
            Err(_) => {
                unreadable.push(Unreadable::IoError(rel_path));
                continue;
            }
        };

        let content = match read_content(path, meta.len()) {
            Ok(c) => c,
            Err(_) => {
                unreadable.push(Unreadable::IoError(rel_path));
                continue;
            }
        };

        let syntax_lang = detect_language(path);

        let (body_html, label, note): (String, String, Option<String>) = match &content {
            Content::Text { content, truncated, total_size } => {
                let body = highlight_code(content, syntax_lang);
                let note = if *truncated {
                    n_truncated += 1;
                    Some(format!(
                        "Truncated preview: showing the first {} bytes out of {} total.",
                        content.len(),
                        total_size
                    ))
                } else {
                    None
                };
                (body, syntax_lang.to_string(), note)
            }
            Content::Binary { preview, total_size, truncated } => {
                n_binary += 1;
                let dump = hex_dump(preview, HEXDUMP_BYTES);
                let shown = preview.len().min(HEXDUMP_BYTES);
                let note = Some(format!(
                    "Binary file: {} bytes total. Showing the first {} bytes as hex.{}",
                    total_size,
                    shown,
                    if *truncated { " (only the first bytes of the file were read)" } else { "" }
                ));
                (escape_html(&dump), "binary".to_string(), note)
            }
        };

        let mut out_rel = rel_path.clone();
        let new_name = format!(
            "{}.html",
            out_rel.file_name().unwrap_or_default().to_string_lossy()
        );
        out_rel.set_file_name(new_name);

        let out_path = output.join(&out_rel);
        if let Some(parent) = out_path.parent() {
            let _ = fs::create_dir_all(parent);
        }

        let depth = out_rel.components().count().saturating_sub(1);
        let back_to_index = "../".repeat(depth);

        let html = render_file_html(
            &rel_path.to_string_lossy(),
            &label,
            &body_html,
            note.as_deref(),
            &back_to_index,
        );

        if fs::write(&out_path, html).is_ok() {
            results.push(FileResult {
                rel_path,
                out_rel,
                label,
            });
        }
    }

    // Build the tree for the index
    let mut root = Node::default();
    for r in &results {
        let components: Vec<String> = r
            .rel_path
            .parent()
            .map(|p| {
                p.components()
                    .map(|c| c.as_os_str().to_string_lossy().to_string())
                    .filter(|s| !s.is_empty())
                    .collect()
            })
            .unwrap_or_default();
        let display_name = r
            .rel_path
            .file_name()
            .unwrap_or_default()
            .to_string_lossy()
            .to_string();
        root.insert_file(
            &components,
            display_name,
            r.label.clone(),
            r.out_rel.clone(),
        );
    }

    let mut tree_html = String::new();
    root.render(&mut tree_html);

    let index_html = render_index_html(&tree_html, results.len(), n_binary, n_truncated);
    let _ = fs::write(output.join("index.html"), index_html);

    let elapsed = start.elapsed();
    println!(
        "Done in {:.2?}: {} files included ({} binary, {} with truncated preview), {} unreadable.",
        elapsed,
        results.len(),
        n_binary,
        n_truncated,
        unreadable.len()
    );
    for u in &unreadable {
        match u {
            Unreadable::IoError(p) => println!("  unreadable (I/O error): {}", p.display()),
        }
    }
    println!("Open {}/index.html in your browser.", output.display());
}

/// Produces a hex dump (in the style of `xxd -C`) of the first `limit` bytes.
fn hex_dump(bytes: &[u8], limit: usize) -> String {
    let show = &bytes[..bytes.len().min(limit)];
    let mut out = String::with_capacity(show.len() * 4);
    for (i, chunk) in show.chunks(16).enumerate() {
        let offset = i * 16;
        out.push_str(&format!("{:08x}  ", offset));
        for (j, b) in chunk.iter().enumerate() {
            out.push_str(&format!("{:02x} ", b));
            if j == 7 {
                out.push(' ');
            }
        }
        let missing = 16usize.saturating_sub(chunk.len());
        for _ in 0..missing {
            out.push_str("   ");
        }
        out.push_str(" |");
        for b in chunk {
            let c = *b;
            if (0x20..0x7f).contains(&c) {
                out.push(c as char);
            } else {
                out.push('.');
            }
        }
        out.push_str("|\n");
    }
    out
}

/// Determines the language (used to pick syntax-highlighting rules)
/// from the file's name/extension. This is only used when the content
/// turns out to actually be text; binary files are classified as such
/// regardless by read_content.
fn detect_language(path: &Path) -> &'static str {
    let file_name = path
        .file_name()
        .map(|s| s.to_string_lossy().to_lowercase())
        .unwrap_or_default();

    match file_name.as_str() {
        "dockerfile" => return "dockerfile",
        "makefile" | "gnumakefile" => return "makefile",
        "cmakelists.txt" => return "cmake",
        ".gitignore" | ".dockerignore" | ".npmignore" => return "plaintext",
        "cargo.toml" | "cargo.lock" => return "ini",
        _ => {}
    }

    let ext = path
        .extension()
        .map(|s| s.to_string_lossy().to_lowercase())
        .unwrap_or_default();

    match ext.as_str() {
        "py" | "pyw" | "pyi" => "python",
        "rs" => "rust",
        "js" | "mjs" | "cjs" => "javascript",
        "jsx" => "javascript",
        "ts" => "typescript",
        "tsx" => "typescript",
        "java" => "java",
        "c" => "c",
        "h" => "c",
        "hpp" | "hh" | "hxx" => "cpp",
        "cpp" | "cc" | "cxx" => "cpp",
        "cs" => "csharp",
        "go" => "go",
        "rb" => "ruby",
        "php" => "php",
        "html" | "htm" => "xml",
        "css" => "css",
        "scss" | "sass" => "scss",
        "less" => "less",
        "json" => "json",
        "xml" => "xml",
        "yaml" | "yml" => "yaml",
        "toml" | "ini" | "cfg" | "conf" => "ini",
        "sh" | "bash" | "zsh" => "bash",
        "sql" => "sql",
        "md" | "markdown" => "markdown",
        "kt" | "kts" => "kotlin",
        "swift" => "swift",
        "lua" => "lua",
        "r" => "r",
        "pl" | "pm" => "perl",
        "groovy" | "gradle" => "groovy",
        "vb" => "vbnet",
        "graphql" | "gql" => "graphql",
        "diff" | "patch" => "diff",
        "bat" | "cmd" => "dos",
        "ps1" => "powershell",
        "vue" => "xml",
        "txt" => "plaintext",
        _ => "plaintext",
    }
}

fn escape_html(s: &str) -> String {
    let mut out = String::with_capacity(s.len());
    for c in s.chars() {
        match c {
            '&' => out.push_str("&amp;"),
            '<' => out.push_str("&lt;"),
            '>' => out.push_str("&gt;"),
            '"' => out.push_str("&quot;"),
            '\'' => out.push_str("&#39;"),
            _ => out.push(c),
        }
    }
    out
}

const STYLE: &str = r#"
:root {
  --bg: #0d1117;
  --panel: #161b22;
  --text: #c9d1d9;
  --muted: #8b949e;
  --accent: #58a6ff;
  --border: #30363d;
}
* { box-sizing: border-box; }
body {
  margin: 0;
  background: var(--bg);
  color: var(--text);
  font-family: -apple-system, BlinkMacSystemFont, "Segoe UI", Helvetica, Arial, sans-serif;
}
header {
  padding: 14px 20px;
  background: var(--panel);
  border-bottom: 1px solid var(--border);
  display: flex;
  align-items: center;
  gap: 12px;
  flex-wrap: wrap;
}
header a { color: var(--accent); text-decoration: none; font-weight: 600; }
header a:hover { text-decoration: underline; }
.path { color: var(--muted); font-size: 0.9em; word-break: break-all; }
.note {
  padding: 10px 20px;
  background: #21262d;
  border-bottom: 1px solid var(--border);
  color: var(--muted);
  font-size: 0.85em;
}
main { padding: 0; }
pre {
  margin: 0;
  padding: 16px 20px 40px 20px;
  overflow-x: auto;
  font-size: 13px;
  line-height: 1.5;
}
code { font-family: "SFMono-Regular", Consolas, "Liberation Mono", Menlo, monospace; }
.container { max-width: 1000px; margin: 30px auto; padding: 0 20px 60px; }
h1 { font-size: 1.6em; }
.meta { color: var(--muted); font-size: 0.9em; margin-bottom: 20px; }
ul.tree { list-style: none; padding-left: 18px; margin: 6px 0; }
ul.tree > li { margin: 4px 0; }
.dir-name { font-weight: 600; color: var(--text); }
.file a { color: var(--accent); text-decoration: none; }
.file a:hover { text-decoration: underline; }
.badge {
  display: inline-block;
  font-size: 0.72em;
  color: var(--muted);
  border: 1px solid var(--border);
  border-radius: 10px;
  padding: 1px 8px;
  margin-left: 6px;
}
.skipped { color: var(--muted); font-size: 0.85em; margin-top: 30px; }

/* Syntax highlighting: generated in Rust, no JavaScript involved. */
.c { color: #8b949e; font-style: italic; }
.s { color: #a5d6ff; }
.n { color: #79c0ff; }
.k { color: #ff7b72; font-weight: 600; }
"#;

fn render_file_html(rel_path: &str, label: &str, body_html: &str, note: Option<&str>, back_to_index: &str) -> String {
    let title = escape_html(rel_path);
    let note_html = match note {
        Some(n) => format!("<div class=\"note\">{}</div>\n", escape_html(n)),
        None => String::new(),
    };
    format!(
        r#"<!DOCTYPE html>
<html lang="en">
<head>
<meta charset="UTF-8">
<title>{title}</title>
<style>{STYLE}</style>
</head>
<body>
<header>
  <a href="{back_to_index}index.html">&larr; Index</a>
  <span class="path">{title}</span>
  <span class="badge">{label}</span>
</header>
{note_html}<main>
<pre><code>{body_html}</code></pre>
</main>
</body>
</html>
"#
    )
}

fn render_index_html(tree_html: &str, n_ok: usize, n_binary: usize, n_truncated: usize) -> String {
    let mut extra = String::new();
    if n_binary > 0 {
        extra.push_str(&format!(
            "<p class=\"skipped\">{} binary files included with a hex-dump preview.</p>",
            n_binary
        ));
    }
    if n_truncated > 0 {
        extra.push_str(&format!(
            "<p class=\"skipped\">{} very large files shown as a truncated preview only.</p>",
            n_truncated
        ));
    }

    let folder_emoji = '\u{1F4C2}';
    format!(
        r#"<!DOCTYPE html>
<html lang="en">
<head>
<meta charset="UTF-8">
<title>Source index</title>
<style>{STYLE}</style>
</head>
<body>
<div class="container">
<h1>{folder_emoji} Source index</h1>
<p class="meta">{n_ok} files included in total.</p>
{tree_html}
{extra}
</div>
</body>
</html>
"#
    )
}