diff --git a/Cargo.lock b/Cargo.lock index 3472150..a45069e 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -230,6 +230,12 @@ dependencies = [ "syn 2.0.119", ] +[[package]] +name = "equivalent" +version = "1.0.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "877a4ace8713b0bcf2a4e7eec82529c029f1d0619886d18145fea96c3ffe5c0f" + [[package]] name = "errno" version = "0.3.14" @@ -334,12 +340,28 @@ version = "0.32.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e629b9b98ef3dd8afe6ca2bd0f89306cec16d43d907889945bc5d6687f2f13c7" +[[package]] +name = "hashbrown" +version = "0.17.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a" + [[package]] name = "heck" version = "0.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea" +[[package]] +name = "indexmap" +version = "2.14.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cc4e190f5d26ca7051642629da2c52fc03bde85a03197c99408dcd291734c855" +dependencies = [ + "equivalent", + "hashbrown", +] + [[package]] name = "inotify" version = "0.9.6" @@ -590,9 +612,11 @@ dependencies = [ "org-ast", "org-parser", "parking_lot", + "serde", "tempfile", "thiserror", "tokio", + "toml", "tracing", "tracing-subscriber", "uuid", @@ -769,6 +793,15 @@ dependencies = [ "syn 3.0.6", ] +[[package]] +name = "serde_spanned" +version = "0.6.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bf41e0cfaf7226dca15e8197172c295a782857fcb97fad1808a166870dee75a3" +dependencies = [ + "serde", +] + [[package]] name = "sharded-slab" version = "0.1.7" @@ -937,6 +970,47 @@ dependencies = [ "syn 3.0.6", ] +[[package]] +name = "toml" +version = "0.8.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "dc1beb996b9d83529a9e75c17a1686767d148d70663143c7854d8b4a09ced362" +dependencies = [ + "serde", + "serde_spanned", + "toml_datetime", + "toml_edit", +] + +[[package]] +name = "toml_datetime" +version = "0.6.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "22cddaf88f4fbc13c51aebbf5f8eceb5c7c5a9da2ac40a13519eb5b0a0e8f11c" +dependencies = [ + "serde", +] + +[[package]] +name = "toml_edit" +version = "0.22.27" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "41fe8c660ae4257887cf66394862d21dbca4a6ddd26f04a3560410406a2f819a" +dependencies = [ + "indexmap", + "serde", + "serde_spanned", + "toml_datetime", + "toml_write", + "winnow", +] + +[[package]] +name = "toml_write" +version = "0.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5d99f8c9a7727884afe522e9bd5edbfc91a3312b36a77b5fb8926e4c31a41801" + [[package]] name = "tracing" version = "0.1.44" @@ -1194,6 +1268,15 @@ version = "0.48.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ed94fce61571a4006852b7389a063ab983c02eb1bb37b47f8272ce92d06d9538" +[[package]] +name = "winnow" +version = "0.7.15" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "df79d97927682d2fd8adb29682d1140b343be4ac0f08fd68b7765d9c059d3945" +dependencies = [ + "memchr", +] + [[package]] name = "yansi" version = "1.0.1" diff --git a/crates/org-parser/src/convert.rs b/crates/org-parser/src/convert.rs index 9da4b40..a4f1e39 100644 --- a/crates/org-parser/src/convert.rs +++ b/crates/org-parser/src/convert.rs @@ -378,6 +378,18 @@ fn convert_element(source: &str, node: Node) -> ParseResult> { preserve_indent: false, number_lines: None, }), + "example" => ElementKind::ExampleBlock(org_ast::ExampleBlock { + contents, + preserve_indent: false, + number_lines: None, + }), + "comment" => ElementKind::CommentBlock(org_ast::CommentBlock { contents }), + "export" => ElementKind::ExportBlock(org_ast::ExportBlock { + backend: params.first().cloned().unwrap_or_default(), + contents, + }), + // For blocks requiring nested element parsing (quote, verse, center, special), + // we'd need recursive parsing. For now, treat as raw SourceBlock placeholder. _ => ElementKind::SourceBlock(SourceBlock { language: Some(name), arguments, diff --git a/crates/ork-server/Cargo.toml b/crates/ork-server/Cargo.toml index 48aa6f4..7ea89ce 100644 --- a/crates/ork-server/Cargo.toml +++ b/crates/ork-server/Cargo.toml @@ -24,6 +24,10 @@ tokio = { version = "1", features = ["full"] } thiserror = "2" anyhow = "1" +# Config parsing +toml = "0.8" +serde = { version = "1", features = ["derive"] } + # Logging tracing = "0.1" tracing-subscriber = { version = "0.3", features = ["env-filter"] } diff --git a/crates/ork-server/src/languages.rs b/crates/ork-server/src/languages.rs new file mode 100644 index 0000000..1e4397f --- /dev/null +++ b/crates/ork-server/src/languages.rs @@ -0,0 +1,282 @@ +//! Language plugin system for code execution. +//! +//! Languages are defined by configuration files that specify how to invoke +//! interpreters. Users add language support via TOML config files in: +//! - ~/.config/orkmode/languages/ +//! - /etc/orkmode/languages/ + +use serde::{Deserialize, Serialize}; +use std::collections::HashMap; +use std::fs; +use std::io::{self, Result}; +use std::path::{Path, PathBuf}; + +/// Defines how to execute code in a specific language. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct LanguageConfig { + /// Command to run (e.g., "python3", "bash"). + pub command: String, + + /// Argument template. Use `{code}` for inline code, `{file}` for temp file path. + /// Examples: ["-c", "{code}"] or ["{file}"] + pub args: Vec, + + /// How to pass code to the interpreter. + #[serde(default)] + pub mode: ExecMode, + + /// File extension for file mode (e.g., ".py", ".sh"). + #[serde(default)] + pub extension: Option, + + /// Alternative names for this language (e.g., ["python", "python3", "py"]). + #[serde(default)] + pub aliases: Vec, +} + +/// How code is passed to the interpreter. +#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq)] +#[serde(rename_all = "lowercase")] +pub enum ExecMode { + /// Pass code as command-line argument (e.g., `bash -c "code"`). + #[default] + Inline, + /// Write code to temp file and pass path (e.g., `python script.py`). + File, + /// Pass code via stdin. + Stdin, +} + +/// Registry of language configurations. +#[derive(Debug, Default)] +pub struct LanguageRegistry { + languages: HashMap, +} + +impl LanguageRegistry { + /// Create a registry with built-in language defaults. + pub fn new() -> Self { + let mut reg = Self::default(); + reg.register_builtins(); + reg + } + + /// Register built-in language defaults. + /// These can be overridden by user config files. + fn register_builtins(&mut self) { + // Shell languages + self.register("bash", LanguageConfig { + command: "bash".into(), + args: vec!["-c".into(), "{code}".into()], + mode: ExecMode::Inline, + extension: Some(".sh".into()), + aliases: vec!["sh".into(), "shell".into()], + }); + + // Python + self.register("python", LanguageConfig { + command: "python3".into(), + args: vec!["-c".into(), "{code}".into()], + mode: ExecMode::Inline, + extension: Some(".py".into()), + aliases: vec!["python3".into(), "py".into()], + }); + + // Ruby + self.register("ruby", LanguageConfig { + command: "ruby".into(), + args: vec!["-e".into(), "{code}".into()], + mode: ExecMode::Inline, + extension: Some(".rb".into()), + aliases: vec!["rb".into()], + }); + + // Perl + self.register("perl", LanguageConfig { + command: "perl".into(), + args: vec!["-e".into(), "{code}".into()], + mode: ExecMode::Inline, + extension: Some(".pl".into()), + aliases: vec!["pl".into()], + }); + + // JavaScript/Node.js + self.register("javascript", LanguageConfig { + command: "node".into(), + args: vec!["-e".into(), "{code}".into()], + mode: ExecMode::Inline, + extension: Some(".js".into()), + aliases: vec!["node".into(), "js".into()], + }); + } + + /// Register a language configuration. + pub fn register(&mut self, name: &str, config: LanguageConfig) { + // Register under primary name + self.languages.insert(name.to_lowercase(), config.clone()); + + // Register under aliases + for alias in &config.aliases { + self.languages.insert(alias.to_lowercase(), config.clone()); + } + } + + /// Look up a language by name or alias. + pub fn get(&self, name: &str) -> Option<&LanguageConfig> { + self.languages.get(&name.to_lowercase()) + } + + /// List all registered language names. + pub fn list(&self) -> Vec<&str> { + let mut names: Vec<_> = self.languages.keys().map(|s| s.as_str()).collect(); + names.sort(); + names + } + + /// Check if a language is registered. + pub fn supports(&self, name: &str) -> bool { + self.languages.contains_key(&name.to_lowercase()) + } + + /// Load language definitions from a directory. + pub fn load_from_dir(&mut self, dir: &Path) -> Result { + if !dir.exists() { + return Ok(0); + } + + let mut count = 0; + for entry in fs::read_dir(dir)? { + let entry = entry?; + let path = entry.path(); + + if path.extension().map(|e| e == "toml").unwrap_or(false) { + match self.load_from_file(&path) { + Ok(_) => count += 1, + Err(e) => eprintln!("Warning: failed to load {:?}: {}", path, e), + } + } + } + + Ok(count) + } + + /// Load a single language definition file. + fn load_from_file(&mut self, path: &Path) -> Result<()> { + let content = fs::read_to_string(path)?; + let config: LanguageConfig = toml::from_str(&content) + .map_err(|e| io::Error::new(io::ErrorKind::InvalidData, e.to_string()))?; + + // Use filename (without extension) as language name + let name = path.file_stem() + .and_then(|s| s.to_str()) + .ok_or_else(|| io::Error::new(io::ErrorKind::InvalidInput, "invalid filename"))?; + + self.register(name, config); + Ok(()) + } + + /// Load languages from standard config directories. + /// Returns number of languages loaded. + pub fn load_standard_configs(&mut self) -> Result { + let mut total = 0; + + // System config (loaded first, can be overridden by user) + total += self.load_from_dir(Path::new("/etc/orkmode/languages"))?; + + // User config + if let Some(home) = std::env::var_os("HOME") { + let user_dir = PathBuf::from(home).join(".config/orkmode/languages"); + total += self.load_from_dir(&user_dir)?; + } + + Ok(total) + } + + /// Build command and arguments for executing code. + pub fn build_command(&self, lang: &str, code: &str) -> Result<(String, Vec)> { + let config = self.get(lang) + .ok_or_else(|| io::Error::new( + io::ErrorKind::InvalidInput, + format!("unsupported language: {} (no plugin found)", lang), + ))?; + + let cmd = config.command.clone(); + let args: Vec = config.args.iter() + .map(|arg| arg.replace("{code}", code)) + .collect(); + + Ok((cmd, args)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use std::io::Write; + + #[test] + fn test_builtin_languages() { + let reg = LanguageRegistry::new(); + // Builtins are registered by default + assert!(reg.get("bash").is_some()); + assert!(reg.get("sh").is_some()); // alias + assert!(reg.get("python").is_some()); + assert!(reg.get("ruby").is_some()); + assert!(reg.get("perl").is_some()); + assert!(reg.get("javascript").is_some()); + assert!(reg.get("node").is_some()); // alias + + // Unknown language still fails + assert!(reg.get("fortran").is_none()); + assert!(reg.build_command("fortran", "code").is_err()); + } + + #[test] + fn test_register_language_override() { + let mut reg = LanguageRegistry::new(); + // Override builtin with custom config + reg.register("bash", LanguageConfig { + command: "/custom/bash".into(), + args: vec!["-c".into(), "{code}".into()], + mode: ExecMode::Inline, + extension: Some(".sh".into()), + aliases: vec!["sh".into()], + }); + + let config = reg.get("bash").unwrap(); + assert_eq!(config.command, "/custom/bash"); // overridden + assert!(reg.get("sh").is_some()); // alias + assert!(reg.get("BASH").is_some()); // case insensitive + } + + #[test] + fn test_build_command_builtin() { + let reg = LanguageRegistry::new(); + // Python is a builtin + let (cmd, args) = reg.build_command("python", "print(1)").unwrap(); + assert_eq!(cmd, "python3"); + assert_eq!(args, vec!["-c", "print(1)"]); + } + + #[test] + fn test_load_from_file() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("rust.toml"); + + let mut file = fs::File::create(&path).unwrap(); + writeln!(file, r#" +command = "rustc" +args = ["--edition", "2021", "-o", "/tmp/out", "{{file}}"] +mode = "file" +extension = ".rs" +aliases = ["rs"] +"#).unwrap(); + + let mut reg = LanguageRegistry::new(); + reg.load_from_file(&path).unwrap(); + + let config = reg.get("rust").unwrap(); + assert_eq!(config.command, "rustc"); + assert!(reg.get("rs").is_some()); // alias + } +} diff --git a/crates/ork-server/src/lib.rs b/crates/ork-server/src/lib.rs index dd9a365..8598f5c 100644 --- a/crates/ork-server/src/lib.rs +++ b/crates/ork-server/src/lib.rs @@ -7,6 +7,8 @@ pub mod state; pub mod namespace; pub mod p9; pub mod sandbox; +pub mod languages; pub use state::OrkState; pub use p9::Server; +pub use languages::LanguageRegistry; diff --git a/crates/ork-server/src/state.rs b/crates/ork-server/src/state.rs index b0baea6..562fd19 100644 --- a/crates/ork-server/src/state.rs +++ b/crates/ork-server/src/state.rs @@ -69,6 +69,9 @@ pub struct OrkState { /// Parser instance. parser: RwLock, + + /// Language registry for code execution. + languages: RwLock, } impl OrkState { @@ -78,11 +81,18 @@ impl OrkState { if !org_dir.is_dir() { return Err(io::Error::new(io::ErrorKind::NotADirectory, "not a directory")); } + + // Load language plugins + let mut languages = crate::languages::LanguageRegistry::new(); + if let Err(e) = languages.load_standard_configs() { + eprintln!("Warning: failed to load language configs: {}", e); + } let state = Self { org_dir, docs: RwLock::new(HashMap::new()), parser: RwLock::new(Parser::new()), + languages: RwLock::new(languages), }; state.scan()?; @@ -426,37 +436,61 @@ impl OrkState { /// Insert #+RESULTS: block after a source block. fn insert_results(&self, content: &str, block_span: Span, stdout: &str) -> String { - let end_offset = block_span.end.offset; + let end_offset = block_span.end.offset.min(content.len()); - // Find the end of the line after the block (in case span doesn't include newline) - let insert_pos = content[end_offset..] - .find('\n') - .map(|i| end_offset + i + 1) - .unwrap_or(content.len()); + // The block span end already points to the start of the line following + // the closing #+END_SRC (it includes the block's trailing newline). + // Only advance to the next line if the span happens to end mid-line + // (i.e. not already at a line boundary). + let insert_pos = if end_offset == 0 + || content.as_bytes().get(end_offset.wrapping_sub(1)) == Some(&b'\n') + || end_offset == content.len() + { + end_offset + } else { + content[end_offset..] + .find('\n') + .map(|i| end_offset + i + 1) + .unwrap_or(content.len()) + }; // Check if there's already a #+RESULTS: block after this let after_block = &content[insert_pos..]; let results_prefix = "#+RESULTS:"; - // Look for #+RESULTS: at the start of the next non-empty line - let (replace_start, replace_end) = if let Some(_results_start) = after_block + // Look for #+RESULTS: in the next few lines (skip blank lines) + let maybe_results_line = after_block .lines() - .next() - .filter(|line| line.trim().starts_with(results_prefix)) - { + .take(3) // Check up to 3 lines (allow for blank lines) + .find(|line| !line.trim().is_empty()) + .filter(|line| line.trim().to_uppercase().starts_with(results_prefix)); + + let (replace_start, replace_end) = if maybe_results_line.is_some() { // Find the extent of the existing results block let abs_start = insert_pos; let mut abs_end = abs_start; - // Results block continues until we hit a non-result line - // (lines starting with : or blank lines after #+RESULTS:) + // Track state: are we in results, and are we in an example block? let mut in_results = false; + let mut in_example_block = false; + for line in after_block.lines() { let trimmed = line.trim(); - if trimmed.starts_with(results_prefix) { + let upper = trimmed.to_uppercase(); + + if upper.starts_with(results_prefix) { in_results = true; abs_end += line.len() + 1; // +1 for newline + } else if in_results && upper.starts_with("#+BEGIN_EXAMPLE") { + in_example_block = true; + abs_end += line.len() + 1; + } else if in_example_block { + abs_end += line.len() + 1; + if upper.starts_with("#+END_EXAMPLE") { + in_example_block = false; + } } else if in_results && (trimmed.starts_with(':') || trimmed.is_empty()) { + // Fixed-width format or blank line continuation abs_end += line.len() + 1; } else if in_results { // End of results block @@ -485,18 +519,37 @@ impl OrkState { } } +/// Minimum number of output lines before switching to #+begin_example block. +/// Matches Emacs `org-babel-min-lines-for-block-output` default. +const MIN_LINES_FOR_BLOCK_OUTPUT: usize = 10; + /// Format stdout as org #+RESULTS: block. +/// Uses fixed-width (`: ` prefix) for <10 lines, #+begin_example block for ≥10. fn format_results(stdout: &str) -> String { if stdout.is_empty() { return String::new(); } + let line_count = stdout.lines().count(); let mut result = String::from("#+RESULTS:\n"); - for line in stdout.lines() { - result.push_str(": "); - result.push_str(line); - result.push('\n'); + + if line_count < MIN_LINES_FOR_BLOCK_OUTPUT { + // Fixed-width format: prefix each line with ": " + for line in stdout.lines() { + result.push_str(": "); + result.push_str(line); + result.push('\n'); + } + } else { + // Example block format for larger output + result.push_str("#+begin_example\n"); + result.push_str(stdout); + if !stdout.ends_with('\n') { + result.push('\n'); + } + result.push_str("#+end_example\n"); } + // Add blank line after results for separation result.push('\n'); result @@ -670,21 +723,29 @@ impl OrkState { fn execute_block(&self, block: &SourceBlock) -> Result { let lang = block.language.as_deref().unwrap_or("sh"); - let (cmd, args): (&str, Vec<&str>) = match lang { - "sh" | "bash" | "shell" => ("bash", vec!["-c", &block.contents]), - "python" | "python3" => ("python3", vec!["-c", &block.contents]), - "ruby" => ("ruby", vec!["-e", &block.contents]), - "perl" => ("perl", vec!["-e", &block.contents]), - "node" | "javascript" | "js" => ("node", vec!["-e", &block.contents]), - _ => return Err(io::Error::new( - io::ErrorKind::InvalidInput, - format!("unsupported language: {}", lang), - )), - }; + // Look up language in registry + let languages = self.languages.read(); + let (cmd, args) = languages.build_command(lang, &block.contents)?; + drop(languages); // Release lock before spawning + + // Convert to &str slices for sandbox + let args_refs: Vec<&str> = args.iter().map(|s| s.as_str()).collect(); // Execute in sandbox with Landlock + timeout let config = SandboxConfig::default(); - sandbox::execute_sandboxed(cmd, &args, &block.contents, &config) + sandbox::execute_sandboxed(&cmd, &args_refs, &block.contents, &config) + } + + /// Reload language plugins. + pub fn reload_languages(&self) -> Result { + let mut languages = self.languages.write(); + *languages = crate::languages::LanguageRegistry::new(); + languages.load_standard_configs() + } + + /// List supported languages. + pub fn list_languages(&self) -> Vec { + self.languages.read().list().iter().map(|s| s.to_string()).collect() } } @@ -1956,4 +2017,36 @@ mod tests { let report = state.lint(Some("w")); assert_eq!(report, "w\tok\n"); } + + #[test] + fn test_format_results_empty() { + assert_eq!(format_results(""), ""); + } + + #[test] + fn test_format_results_short() { + // < 10 lines uses fixed-width format (: prefix) + let stdout = "line1\nline2\nline3\n"; + let result = format_results(stdout); + assert!(result.starts_with("#+RESULTS:\n")); + assert!(result.contains(": line1\n")); + assert!(result.contains(": line2\n")); + assert!(result.contains(": line3\n")); + assert!(!result.contains("#+begin_example")); + } + + #[test] + fn test_format_results_long() { + // >= 10 lines uses #+begin_example block + let lines: Vec<_> = (1..=15).map(|i| format!("line{}", i)).collect(); + let stdout = lines.join("\n"); + let result = format_results(&stdout); + assert!(result.starts_with("#+RESULTS:\n")); + assert!(result.contains("#+begin_example\n")); + assert!(result.contains("#+end_example\n")); + assert!(result.contains("line1\n")); + assert!(result.contains("line15")); + // Should NOT have : prefix in example block + assert!(!result.contains(": line1")); + } }