diff --git a/Cargo.lock b/Cargo.lock index b99df352..a2c25e9e 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2,6 +2,18 @@ # It is not intended for manual editing. version = 4 +[[package]] +name = "ahash" +version = "0.8.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5a15f179cd60c4584b8a8c596927aadc462e27f2ca70c04e0071964a73ba7a75" +dependencies = [ + "cfg-if", + "once_cell", + "version_check", + "zerocopy", +] + [[package]] name = "aho-corasick" version = "1.1.4" @@ -56,6 +68,16 @@ dependencies = [ "allocator-api2", ] +[[package]] +name = "cc" +version = "1.4.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0ad534f4357a5264cce5019c989cf66a4f0dc4e0d1b1d15f8aacec0ff7360273" +dependencies = [ + "find-msvc-tools", + "shlex", +] + [[package]] name = "cfg-if" version = "1.0.4" @@ -271,6 +293,18 @@ version = "3.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "dea2df4cf52843e0452895c455a1a2cfbb842a1e7329671acf418fdc53ed4c59" +[[package]] +name = "fallible-iterator" +version = "0.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2acce4a10f12dc2fb14a218589d4f1f62ef011b2d0cc4b3cb1bba8e94da14649" + +[[package]] +name = "fallible-streaming-iterator" +version = "0.1.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7360491ce676a36bf9bb3c56c1aa791658183a54d2744120f27285738d90465a" + [[package]] name = "fd-lock" version = "4.0.4" @@ -282,6 +316,12 @@ dependencies = [ "windows-sys 0.59.0", ] +[[package]] +name = "find-msvc-tools" +version = "0.1.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d45db016d36b838f563236e9193d0ee6ce38f3f68b6c94e914b4929c96bbb890" + [[package]] name = "fnv" version = "1.0.7" @@ -321,6 +361,15 @@ dependencies = [ "stable_deref_trait", ] +[[package]] +name = "hashbrown" +version = "0.14.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e5274423e17b7c9fc20b6e7e208532f9b19825d82dfd615708b70edd83df41f1" +dependencies = [ + "ahash", +] + [[package]] name = "hashbrown" version = "0.15.5" @@ -336,6 +385,15 @@ version = "0.16.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "841d1cc9bed7f9236f321df977030373f4a4163ae1a7dbfe1a51a2c1a51d9100" +[[package]] +name = "hashlink" +version = "0.9.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6ba4ff7128dee98c7dc9794b6a411377e1404dba1c97deb8d1a55297bd25d8af" +dependencies = [ + "hashbrown 0.14.5", +] + [[package]] name = "heck" version = "0.5.0" @@ -385,6 +443,17 @@ version = "0.2.16" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b6d2cec3eae94f9f509c767b45932f1ada8350c4bdb85af2fcab4a3c14807981" +[[package]] +name = "libsqlite3-sys" +version = "0.30.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2e99fb7a497b1e3339bc746195567ed8d3e24945ecd636e3619d20b9de9e9149" +dependencies = [ + "cc", + "pkg-config", + "vcpkg", +] + [[package]] name = "linux-raw-sys" version = "0.12.1" @@ -433,6 +502,12 @@ dependencies = [ "libc", ] +[[package]] +name = "once_cell" +version = "1.21.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" + [[package]] name = "paste" version = "1.0.15" @@ -486,6 +561,7 @@ dependencies = [ "pd-host-function 0.1.0", "regex", "rt-format", + "rusqlite", "rustyline", "self_cell", "serde", @@ -518,6 +594,12 @@ version = "0.2.16" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3b3cff922bd51709b605d9ead9aa71031d81447142d828eb4a6eba76fe619f9b" +[[package]] +name = "pkg-config" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548" + [[package]] name = "proc-macro2" version = "1.0.106" @@ -611,6 +693,20 @@ dependencies = [ "regex", ] +[[package]] +name = "rusqlite" +version = "0.32.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7753b721174eb8ff87a9a0e799e2d7bc3749323e773db92e0984debb00019d6e" +dependencies = [ + "bitflags 2.11.0", + "fallible-iterator", + "fallible-streaming-iterator", + "hashlink", + "libsqlite3-sys", + "smallvec", +] + [[package]] name = "rustc-hash" version = "2.1.1" @@ -708,6 +804,12 @@ dependencies = [ "zmij", ] +[[package]] +name = "shlex" +version = "2.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" + [[package]] name = "smallvec" version = "1.15.1" @@ -782,6 +884,18 @@ version = "0.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821" +[[package]] +name = "vcpkg" +version = "0.2.15" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "accd4ea62f7bb7a82fe23066fb0957d48ef677f6eeb8215f372f52e48bb32426" + +[[package]] +name = "version_check" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" + [[package]] name = "wasmtime-internal-core" version = "42.0.1" @@ -900,6 +1014,26 @@ version = "0.52.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "589f6da84c646204747d1270a2a5661ea66ed1cced2631d546fdfb155959f9ec" +[[package]] +name = "zerocopy" +version = "0.8.56" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "556764e583adb45a9f8d413c2a147fa7e8d821e48e12b14fd560b607998b75eb" +dependencies = [ + "zerocopy-derive", +] + +[[package]] +name = "zerocopy-derive" +version = "0.8.56" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f2ab42fc20575779bd240faa45f94a74256f755c0fa9e89f0ede20d91d0cdfc1" +dependencies = [ + "proc-macro2", + "quote", + "syn", +] + [[package]] name = "zmij" version = "1.0.21" diff --git a/Cargo.toml b/Cargo.toml index 5cfc571d..c59d0a0e 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -27,6 +27,7 @@ name = "vm" [features] default = ["runtime", "cli", "cranelift-jit"] runtime = [] +sqlite = ["runtime", "dep:rusqlite"] edge-abi = [ "dep:edge_abi", "edge_abi/console", @@ -60,6 +61,7 @@ cranelift-jit = { version = "0.129.1", optional = true } cranelift-module = { version = "0.129.1", optional = true } cranelift-native = { version = "0.129.1", optional = true } pd-host-function = { path = "./pd-host-function", version = "0.1.0" } +rusqlite = { version = "0.32", default-features = false, features = ["bundled", "hooks", "limits"], optional = true } edge_abi = { package = "pd-edge-abi", version = "0.1.1", default-features = false, optional = true } futures-channel = "0.3" paste = "1" diff --git a/build.rs b/build.rs index 09cfa5ac..ccc27766 100644 --- a/build.rs +++ b/build.rs @@ -69,6 +69,10 @@ impl HostBindingKind { /// Documented call-index blocks shared by builtins and host imports. /// /// Must match the block table in `src/builtins/catalog.rs`. +/// The ordinary block's top four IDs are frozen for SQLite. Keep allocation +/// explicit here: incrementing a `u16` cursor from `0xFFFF` would overflow. +pub(crate) const SQLITE_RESERVED_TOP_START: u16 = 0xFFFC; +pub(crate) const SQLITE_RESERVED_TOP_END: u16 = u16::MAX; pub(crate) const ORDINARY_BLOCK_START: u16 = 0xFFA2; pub(crate) const SPECIAL_CALL_BLOCK_START: u16 = 0xFF90; pub(crate) const SPECIAL_CALL_BLOCK_END: u16 = 0xFFA1; @@ -143,11 +147,24 @@ fn main() { .join("runtime") .join("namespaces.rs"); println!("cargo:rerun-if-changed={}", namespace_manifest.display()); - let namespaces = parse_namespace_manifest(&namespace_manifest); + let mut namespaces = parse_namespace_manifest(&namespace_manifest); let catalog_path = manifest_dir.join("src").join("builtins").join("catalog.rs"); println!("cargo:rerun-if-changed={}", catalog_path.display()); - let catalog = parse_catalog(&catalog_path); + let mut catalog = parse_catalog(&catalog_path); + + // The SQLite namespace is optional: its builtin module links rusqlite, + // which is not available on every target or without the `sqlite` feature. + // When the feature is off (or the target is wasm32, where rusqlite's + // bundled build is unsupported), drop the namespace and its static + // catalog IDs so the generated catalog, dispatch, and compiler namespace + // surface stay consistent and feature-clean. + let sqlite_enabled = env::var_os("CARGO_FEATURE_SQLITE").is_some() + && env::var("CARGO_CFG_TARGET_ARCH").as_deref() != Ok("wasm32"); + if !sqlite_enabled { + namespaces.retain(|namespace| namespace.namespace != "sqlite"); + catalog.retain(|entry| !entry.source_name.starts_with("sqlite::")); + } let host_sources = [SourceSpec { path: "src/builtins/runtime/host.rs".to_string(), @@ -229,10 +246,20 @@ fn write_generated_file(path: &Path, contents: &str) { fn builtin_source_specs(namespaces: &[NamespaceDecl]) -> Vec { namespaces .iter() - .map(|namespace| SourceSpec { - path: format!("src/builtins/runtime/{}.rs", namespace.module), - module: namespace.module.clone(), - category: SourceCategory::NamespacedBuiltin, + .map(|namespace| { + // At this layer the `io` namespace maps directly to the blocking + // backend; the async/blocking split is introduced later with the + // host async execution layer. + let path = if namespace.module == "io" { + "src/builtins/runtime/io/blocking.rs".to_string() + } else { + format!("src/builtins/runtime/{}.rs", namespace.module) + }; + SourceSpec { + path, + module: namespace.module.clone(), + category: SourceCategory::NamespacedBuiltin, + } }) .collect() } @@ -522,6 +549,7 @@ fn strip_quoted(value: &str) -> Option { /// - a catalog variant does not match the derived variant for its source name; /// - a class disagrees with the dispatch classification (ordinary vs /// special-call) or with the `__` internal-name prefix; +/// - a non-SQLite entry uses one of the frozen top-u16 SQLite IDs; /// - an ID falls outside its documented block. pub(crate) fn validate_catalog_contract( entries: &[CatalogEntry], @@ -554,6 +582,16 @@ pub(crate) fn validate_catalog_contract( entry.source_name, entry.variant ); } + if (SQLITE_RESERVED_TOP_START..=SQLITE_RESERVED_TOP_END).contains(&entry.id) + && !entry.source_name.starts_with("sqlite::") + { + panic!( + "builtin '{}' id 0x{:04X} falls in the SQLite-reserved top-u16 range \ + 0x{SQLITE_RESERVED_TOP_START:04X}..=0x{SQLITE_RESERVED_TOP_END:04X}; \ + do not allocate IDs by arithmetic", + entry.source_name, entry.id + ); + } let is_special_call = special_variants.contains(&entry.variant); match entry.class { CatalogClass::Ordinary => { @@ -766,6 +804,11 @@ fn render_builtin_catalog( .collect::>(), )); + writeln!( + &mut out, + "// The top-u16 range 0xFFFC..=0xFFFF is reserved for SQLite's frozen IDs; do not allocate it arithmetically." + ) + .unwrap(); writeln!( &mut out, "#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]" diff --git a/crates/rustscript/Cargo.toml b/crates/rustscript/Cargo.toml index 6279a1e1..66b33b20 100644 --- a/crates/rustscript/Cargo.toml +++ b/crates/rustscript/Cargo.toml @@ -13,6 +13,7 @@ name = "rustscript" [features] default = ["runtime", "cli", "cranelift-jit"] runtime = ["pd_vm_crate/runtime"] +sqlite = ["pd_vm_crate/sqlite"] edge-abi = ["pd_vm_crate/edge-abi"] cli = ["pd_vm_crate/cli"] cranelift-jit = ["pd_vm_crate/cranelift-jit"] diff --git a/pd-host-function/src/lib.rs b/pd-host-function/src/lib.rs index 4aa1d3bc..5c4abaeb 100644 --- a/pd-host-function/src/lib.rs +++ b/pd-host-function/src/lib.rs @@ -223,13 +223,13 @@ fn generate_vm_wrapper( Ok(quote! { #[allow(dead_code)] - pub(super) fn #wrapper_name(#(#imm_wrapper_params),*) -> #wrapper_output { + pub(crate) fn #wrapper_name(#(#imm_wrapper_params),*) -> #wrapper_output { #(#imm_extract_stmts)* #call_expr } #[allow(dead_code)] - pub(super) fn #mutable_wrapper_name(#(#mut_wrapper_params),*) -> #wrapper_output { + pub(crate) fn #mutable_wrapper_name(#(#mut_wrapper_params),*) -> #wrapper_output { #(#mut_extract_stmts)* #call_expr } diff --git a/pd-vm-nostd/src/generated_builtin_ids.rs b/pd-vm-nostd/src/generated_builtin_ids.rs index bc078a5a..d779e5de 100644 --- a/pd-vm-nostd/src/generated_builtin_ids.rs +++ b/pd-vm-nostd/src/generated_builtin_ids.rs @@ -5,6 +5,9 @@ // pd-vm-nostd dispatches on the same static indices without a build script. // The workspace test `static_builtin_ids_are_frozen` fails when this file // drifts from the catalog; do not edit by hand. +// +// The top-u16 range 0xFFFC..=0xFFFF is reserved for SQLite's frozen IDs; +// never allocate a new builtin there by incrementing an integer cursor. #![allow(dead_code)] @@ -50,6 +53,11 @@ pub const IO_WRITE_CALL_INDEX: u16 = 0xFFB9; pub const IO_FLUSH_CALL_INDEX: u16 = 0xFFBA; pub const IO_CLOSE_CALL_INDEX: u16 = 0xFFBB; pub const IO_EXISTS_CALL_INDEX: u16 = 0xFFBC; +pub const SQLITE_OPEN_CALL_INDEX: u16 = 0xFFC3; +pub const SQLITE_EXECUTE_CALL_INDEX: u16 = 0xFFFC; +pub const SQLITE_QUERY_CALL_INDEX: u16 = 0xFFFD; +pub const SQLITE_TRANSACTION_CALL_INDEX: u16 = 0xFFFE; +pub const SQLITE_CLOSE_CALL_INDEX: u16 = 0xFFFF; pub const RE_MATCH_CALL_INDEX: u16 = 0xFFBD; pub const RE_FIND_CALL_INDEX: u16 = 0xFFBE; pub const RE_REPLACE_CALL_INDEX: u16 = 0xFFBF; @@ -163,6 +171,7 @@ pub const ALL_CALL_INDICES: &[u16] = &[ RE_SPLIT_CALL_INDEX, RE_CAPTURES_CALL_INDEX, JSON_ENCODE_CALL_INDEX, + SQLITE_OPEN_CALL_INDEX, JSON_DECODE_CALL_INDEX, JIT_SET_CONFIG_CALL_INDEX, JIT_GET_CONFIG_CALL_INDEX, @@ -219,4 +228,8 @@ pub const ALL_CALL_INDICES: &[u16] = &[ MATH_CLAMP_CALL_INDEX, MATH_MUL_ADD_CALL_INDEX, COUNT_CALL_INDEX, + SQLITE_EXECUTE_CALL_INDEX, + SQLITE_QUERY_CALL_INDEX, + SQLITE_TRANSACTION_CALL_INDEX, + SQLITE_CLOSE_CALL_INDEX, ]; diff --git a/src/builtins/catalog.rs b/src/builtins/catalog.rs index 93bf4c2a..0426e250 100644 --- a/src/builtins/catalog.rs +++ b/src/builtins/catalog.rs @@ -63,6 +63,11 @@ builtin_id!(0xFFB9, "io::write", IoWrite, Ordinary, none); builtin_id!(0xFFBA, "io::flush", IoFlush, Ordinary, none); builtin_id!(0xFFBB, "io::close", IoClose, Ordinary, none); builtin_id!(0xFFBC, "io::exists", IoExists, Ordinary, none); +builtin_id!(0xFFC3, "sqlite::open", SqliteOpen, Ordinary, none); +builtin_id!(0xFFFC, "sqlite::execute", SqliteExecute, Ordinary, none); +builtin_id!(0xFFFD, "sqlite::query", SqliteQuery, Ordinary, none); +builtin_id!(0xFFFE, "sqlite::transaction", SqliteTransaction, Ordinary, none); +builtin_id!(0xFFFF, "sqlite::close", SqliteClose, Ordinary, none); builtin_id!(0xFFBD, "re::match", ReMatch, Ordinary, none); builtin_id!(0xFFBE, "re::find", ReFind, Ordinary, none); builtin_id!(0xFFBF, "re::replace", ReReplace, Ordinary, none); diff --git a/src/builtins/runtime/io.rs b/src/builtins/runtime/io.rs deleted file mode 100644 index b9ef5340..00000000 --- a/src/builtins/runtime/io.rs +++ /dev/null @@ -1,485 +0,0 @@ -use std::collections::HashMap; -use std::fs::OpenOptions; -use std::future::Future; -use std::io::{Read, Write}; -use std::pin::Pin; -use std::process::{Child, Command, Stdio}; -use std::task::{Context, Poll}; - -use futures_channel::oneshot; -use pd_host_function::pd_host_function; - -use super::HostCallResult; -use crate::vm::{CallReturn, HostOpId, Value, Vm, VmError, VmResult}; - -pub(crate) struct IoState { - pub(super) next_handle: i64, - pub(super) handles: HashMap, - pending_ops: HashMap>, -} - -impl Default for IoState { - fn default() -> Self { - Self { - next_handle: 1, - handles: HashMap::new(), - pending_ops: HashMap::new(), - } - } -} - -pub(super) enum IoHandle { - File(std::fs::File), - PopenRead { child: Child }, - PopenWrite { child: Child }, -} - -struct IoAsyncCompletion { - restored_handle: Option<(i64, IoHandle)>, - result: VmResult, -} - -pub(super) fn cancel_pending_op(vm: &mut Vm, op_id: HostOpId) { - vm.io_state.pending_ops.remove(&op_id); -} - -pub(super) fn poll_builtin_io_op( - vm: &mut Vm, - op_id: HostOpId, - cx: &mut Context<'_>, -) -> Poll> { - let poll_result = { - let receiver = match vm.io_state.pending_ops.get_mut(&op_id) { - Some(receiver) => receiver, - None => { - return Poll::Ready(Err(VmError::HostError(format!( - "unknown builtin io op {op_id}", - )))); - } - }; - Pin::new(receiver).poll(cx) - }; - - match poll_result { - Poll::Pending => Poll::Pending, - Poll::Ready(Ok(completion)) => { - vm.io_state.pending_ops.remove(&op_id); - if let Some((handle_id, handle)) = completion.restored_handle { - vm.io_state.handles.insert(handle_id, handle); - } - Poll::Ready(completion.result) - } - Poll::Ready(Err(_)) => { - vm.io_state.pending_ops.remove(&op_id); - Poll::Ready(Err(VmError::HostError(format!( - "builtin io op {op_id} was cancelled", - )))) - } - } -} - -pub(super) fn close_all_handles(vm: &mut Vm) { - let handles = std::mem::take(&mut vm.io_state.handles); - for (_, handle) in handles { - let _ = close_io_handle(handle); - } -} - -/// Opens a file handle for runtime I/O. -#[pd_host_function(name = "io::open")] -pub(super) fn builtin_io_open( - vm: &mut Vm, - path: &str, - mode: &str, -) -> VmResult> { - let reserved_id = io_reserve_handle_id(vm); - let path = path.to_string(); - let mode = mode.to_string(); - let op_id = schedule_io_task(vm, move || { - let mut options = OpenOptions::new(); - match mode.as_str() { - "r" => { - options.read(true); - } - "w" => { - options.write(true).create(true).truncate(true); - } - "a" => { - options.write(true).create(true).append(true); - } - "r+" => { - options.read(true).write(true); - } - "w+" => { - options.read(true).write(true).create(true).truncate(true); - } - "a+" => { - options.read(true).write(true).create(true).append(true); - } - other => { - return IoAsyncCompletion { - restored_handle: None, - result: Err(VmError::HostError(format!( - "unsupported io_open mode '{other}', expected r/w/a/r+/w+/a+", - ))), - }; - } - } - - match options.open(path) { - Ok(file) => IoAsyncCompletion { - restored_handle: Some((reserved_id, IoHandle::File(file))), - result: Ok(CallReturn::one(Value::Int(reserved_id))), - }, - Err(err) => IoAsyncCompletion { - restored_handle: None, - result: Err(VmError::HostError(format!("io_open failed: {err}"))), - }, - } - })?; - Ok(HostCallResult::Pending(op_id)) -} - -/// Starts a child process and returns a process-backed handle. -#[pd_host_function(name = "io::popen")] -pub(super) fn builtin_io_popen( - vm: &mut Vm, - command: &str, - mode: &str, -) -> VmResult> { - if mode != "r" && mode != "w" { - return Err(VmError::HostError(format!( - "unsupported io_popen mode '{mode}', expected r or w" - ))); - } - let reserved_id = io_reserve_handle_id(vm); - let command = command.to_string(); - let mode = mode.to_string(); - let op_id = schedule_io_task(vm, move || { - let child = match spawn_shell_command(command.as_str(), mode.as_str()) { - Ok(child) => child, - Err(err) => { - return IoAsyncCompletion { - restored_handle: None, - result: Err(err), - }; - } - }; - let handle = match mode.as_str() { - "r" => { - if child.stdout.is_none() { - return IoAsyncCompletion { - restored_handle: None, - result: Err(VmError::HostError( - "io_popen('r') did not provide stdout pipe".to_string(), - )), - }; - } - IoHandle::PopenRead { child } - } - "w" => { - if child.stdin.is_none() { - return IoAsyncCompletion { - restored_handle: None, - result: Err(VmError::HostError( - "io_popen('w') did not provide stdin pipe".to_string(), - )), - }; - } - IoHandle::PopenWrite { child } - } - _ => unreachable!("mode validated above"), - }; - IoAsyncCompletion { - restored_handle: Some((reserved_id, handle)), - result: Ok(CallReturn::one(Value::Int(reserved_id))), - } - })?; - Ok(HostCallResult::Pending(op_id)) -} - -/// Reads all remaining text from an I/O handle. -#[pd_host_function(name = "io::read_all")] -pub(super) fn builtin_io_read_all(vm: &mut Vm, handle_id: i64) -> VmResult> { - let handle = io_take_handle(vm, handle_id)?; - let op_id = schedule_io_task(vm, move || { - let mut handle = handle; - let mut out = String::new(); - let result = match &mut handle { - IoHandle::File(file) => file - .read_to_string(&mut out) - .map_err(|err| VmError::HostError(format!("io_read_all failed: {err}"))) - .map(|_| CallReturn::one(Value::string(out))), - IoHandle::PopenRead { child } => { - let stdout = match child.stdout.as_mut() { - Some(stdout) => stdout, - None => { - return IoAsyncCompletion { - restored_handle: Some((handle_id, handle)), - result: Err(VmError::HostError( - "io_read_all popen handle missing stdout".to_string(), - )), - }; - } - }; - stdout - .read_to_string(&mut out) - .map_err(|err| VmError::HostError(format!("io_read_all failed: {err}"))) - .map(|_| CallReturn::one(Value::string(out))) - } - IoHandle::PopenWrite { .. } => Err(VmError::HostError( - "io_read_all requires a readable handle".to_string(), - )), - }; - IoAsyncCompletion { - restored_handle: Some((handle_id, handle)), - result, - } - })?; - Ok(HostCallResult::Pending(op_id)) -} - -/// Reads a single line of text from an I/O handle. -#[pd_host_function(name = "io::read_line")] -pub(super) fn builtin_io_read_line( - vm: &mut Vm, - handle_id: i64, -) -> VmResult> { - let handle = io_take_handle(vm, handle_id)?; - let op_id = schedule_io_task(vm, move || { - let mut handle = handle; - let result = match &mut handle { - IoHandle::File(file) => { - read_line_from_reader(file).map(|line| CallReturn::one(Value::string(line))) - } - IoHandle::PopenRead { child } => { - let stdout = match child.stdout.as_mut() { - Some(stdout) => stdout, - None => { - return IoAsyncCompletion { - restored_handle: Some((handle_id, handle)), - result: Err(VmError::HostError( - "io_read_line popen handle missing stdout".to_string(), - )), - }; - } - }; - read_line_from_reader(stdout).map(|line| CallReturn::one(Value::string(line))) - } - IoHandle::PopenWrite { .. } => Err(VmError::HostError( - "io_read_line requires a readable handle".to_string(), - )), - }; - IoAsyncCompletion { - restored_handle: Some((handle_id, handle)), - result, - } - })?; - Ok(HostCallResult::Pending(op_id)) -} - -/// Writes text to an I/O handle. -#[pd_host_function(name = "io::write")] -pub(super) fn builtin_io_write( - vm: &mut Vm, - handle_id: i64, - text: &str, -) -> VmResult> { - let bytes = text.as_bytes().to_vec(); - let handle = io_take_handle(vm, handle_id)?; - let op_id = schedule_io_task(vm, move || { - let mut handle = handle; - let result = match &mut handle { - IoHandle::File(file) => file - .write(&bytes) - .map_err(|err| VmError::HostError(format!("io_write failed: {err}"))) - .map(|written| CallReturn::one(Value::Int(written as i64))), - IoHandle::PopenWrite { child } => { - let stdin = match child.stdin.as_mut() { - Some(stdin) => stdin, - None => { - return IoAsyncCompletion { - restored_handle: Some((handle_id, handle)), - result: Err(VmError::HostError( - "io_write popen handle missing stdin".to_string(), - )), - }; - } - }; - stdin - .write(&bytes) - .map_err(|err| VmError::HostError(format!("io_write failed: {err}"))) - .map(|written| CallReturn::one(Value::Int(written as i64))) - } - IoHandle::PopenRead { .. } => Err(VmError::HostError( - "io_write requires a writable handle".to_string(), - )), - }; - IoAsyncCompletion { - restored_handle: Some((handle_id, handle)), - result, - } - })?; - Ok(HostCallResult::Pending(op_id)) -} - -/// Flushes buffered output for an I/O handle. -#[pd_host_function(name = "io::flush")] -pub(super) fn builtin_io_flush(vm: &mut Vm, handle_id: i64) -> VmResult> { - let handle = io_take_handle(vm, handle_id)?; - let op_id = schedule_io_task(vm, move || { - let mut handle = handle; - let result = match &mut handle { - IoHandle::File(file) => file - .flush() - .map_err(|err| VmError::HostError(format!("io_flush failed: {err}"))) - .map(|_| CallReturn::one(Value::Bool(true))), - IoHandle::PopenWrite { child } => { - let stdin = match child.stdin.as_mut() { - Some(stdin) => stdin, - None => { - return IoAsyncCompletion { - restored_handle: Some((handle_id, handle)), - result: Err(VmError::HostError( - "io_flush popen handle missing stdin".to_string(), - )), - }; - } - }; - stdin - .flush() - .map_err(|err| VmError::HostError(format!("io_flush failed: {err}"))) - .map(|_| CallReturn::one(Value::Bool(true))) - } - IoHandle::PopenRead { .. } => Ok(CallReturn::one(Value::Bool(true))), - }; - IoAsyncCompletion { - restored_handle: Some((handle_id, handle)), - result, - } - })?; - Ok(HostCallResult::Pending(op_id)) -} - -/// Closes an I/O handle. -#[pd_host_function(name = "io::close")] -pub(super) fn builtin_io_close(vm: &mut Vm, handle_id: i64) -> VmResult> { - let handle = io_take_handle(vm, handle_id)?; - let op_id = schedule_io_task(vm, move || IoAsyncCompletion { - restored_handle: None, - result: close_io_handle(handle).map(|_| CallReturn::one(Value::Bool(true))), - })?; - Ok(HostCallResult::Pending(op_id)) -} - -/// Returns whether a file system path exists. -#[pd_host_function(name = "io::exists")] -pub(super) fn builtin_io_exists(vm: &mut Vm, path: &str) -> VmResult> { - let path = path.to_string(); - let op_id = schedule_io_task(vm, move || IoAsyncCompletion { - restored_handle: None, - result: Ok(CallReturn::one(Value::Bool( - std::path::Path::new(path.as_str()).exists(), - ))), - })?; - Ok(HostCallResult::Pending(op_id)) -} - -fn spawn_shell_command(command: &str, mode: &str) -> VmResult { - let mut process = if cfg!(windows) { - let mut cmd = Command::new("cmd"); - cmd.arg("/C").arg(command); - cmd - } else { - let mut cmd = Command::new("sh"); - cmd.arg("-c").arg(command); - cmd - }; - - match mode { - "r" => { - process.stdout(Stdio::piped()).stdin(Stdio::null()); - } - "w" => { - process.stdin(Stdio::piped()).stdout(Stdio::null()); - } - _ => {} - } - - process - .spawn() - .map_err(|err| VmError::HostError(format!("io_popen failed: {err}"))) -} - -fn io_reserve_handle_id(vm: &mut Vm) -> i64 { - let id = vm.io_state.next_handle; - vm.io_state.next_handle = vm.io_state.next_handle.saturating_add(1); - id -} - -fn io_take_handle(vm: &mut Vm, handle_id: i64) -> VmResult { - if handle_id <= 0 { - return Err(VmError::HostError(format!( - "invalid io handle id {handle_id}; expected positive handle id" - ))); - } - vm.io_state - .handles - .remove(&handle_id) - .ok_or_else(|| VmError::HostError(format!("io handle {handle_id} not found"))) -} - -fn schedule_io_task( - vm: &mut Vm, - task: impl FnOnce() -> IoAsyncCompletion + Send + 'static, -) -> VmResult { - let op_id = vm.allocate_host_op_id(); - let (sender, receiver) = oneshot::channel(); - std::thread::Builder::new() - .name("pd-vm-io".to_string()) - .spawn(move || { - let completion = task(); - let _ = sender.send(completion); - }) - .map_err(|err| VmError::HostError(format!("failed to spawn io task: {err}")))?; - vm.io_state.pending_ops.insert(op_id, receiver); - Ok(op_id) -} - -fn close_io_handle(mut handle: IoHandle) -> VmResult<()> { - match &mut handle { - IoHandle::File(file) => { - file.flush().ok(); - } - IoHandle::PopenRead { child } => { - child - .wait() - .map_err(|err| VmError::HostError(format!("io_close popen wait failed: {err}")))?; - } - IoHandle::PopenWrite { child } => { - let _ = child.stdin.take(); - child - .wait() - .map_err(|err| VmError::HostError(format!("io_close popen wait failed: {err}")))?; - } - } - Ok(()) -} - -fn read_line_from_reader(reader: &mut impl Read) -> VmResult { - let mut bytes = Vec::new(); - let mut one = [0u8; 1]; - loop { - let read = reader - .read(&mut one) - .map_err(|err| VmError::HostError(format!("io_read_line failed: {err}")))?; - if read == 0 { - break; - } - bytes.push(one[0]); - if one[0] == b'\n' { - break; - } - } - Ok(String::from_utf8_lossy(&bytes).into_owned()) -} diff --git a/src/builtins/runtime/io/blocking.rs b/src/builtins/runtime/io/blocking.rs new file mode 100644 index 00000000..033e79ef --- /dev/null +++ b/src/builtins/runtime/io/blocking.rs @@ -0,0 +1,1209 @@ +use std::collections::HashMap; +use std::fs::OpenOptions; +use std::io::{Read, Write}; +use std::path::{Path, PathBuf}; +use std::process::{Child, Command, Stdio}; +use std::sync::atomic::{AtomicBool, Ordering}; +use std::sync::{Arc, Mutex}; +use std::task::{Context, Poll, Waker}; +use std::thread::JoinHandle; + +use pd_host_function::pd_host_function; + +use super::HostCallResult; +use crate::vm::operation::driver::HostOperation; +use crate::vm::operation::error::{OperationError, OperationErrorCode, OperationResult}; +use crate::vm::operation::reason::OperationCancelReason; +use crate::vm::operation::{OperationId, OperationSpec}; +use crate::vm::resource::close::{CloseProgress, HostResource}; +use crate::vm::resource::error::{ResourceError, ResourceErrorCode, ResourceResult}; +use crate::vm::resource::{ResourceCloseReason, ResourceHandle}; +use crate::vm::{CallReturn, HostOpId, Value, Vm, VmError, VmResult}; + +/// Adapter-declared per-VM IO host state. +/// +/// Live IO handles are typed [`IoResource`]s owned by the VM's execution +/// scope; in-flight IO work is driven by concrete [`HostOperation`] drivers +/// registered in the same scope. This state itself lives in the execution +/// scope's typed arena (accessed lazily through +/// `ExecutionScope::scope_state_or_insert_with`), so it follows the scope +/// lifecycle: it is destroyed on reset/drop and recreated fresh on next use. +/// The only state kept here is the per-op completion mailbox that carries the +/// guest-visible result value from the worker thread back to +/// [`poll_builtin_io_op`]. Polling and cancellation of the operations +/// themselves go directly through the scope's operation registry — this map +/// is a value mailbox, not a poller table. +#[derive(Default)] +pub(crate) struct IoState { + /// Packed [`OperationId::raw`] -> completion mailbox for pending IO ops. + pending_results: HashMap>, +} + +/// A file / child-process backed IO handle. +pub(super) enum IoHandle { + File(std::fs::File), + PopenRead { child: Child }, + PopenWrite { child: Child }, +} + +/// The typed resource stored in the execution scope for one IO handle. +/// +/// The handle lives behind an `Arc>>` so a worker thread +/// performing read/write/flush/close can transiently take the handle while +/// the resource itself stays in the scope table. Closing is exact-once: the +/// first close (via `io::close` worker or the generic scope close) takes the +/// handle and releases the OS resource. +struct IoResource { + handle: Arc>>, + closed: Arc, +} + +impl IoResource { + fn new(handle: IoHandle) -> Self { + Self { + handle: Arc::new(Mutex::new(Some(handle))), + closed: Arc::new(AtomicBool::new(false)), + } + } + + /// Takes the inner handle for a worker thread (exact-once per close). + fn take_handle(&self) -> Option { + self.handle + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) + .take() + } + + /// Restores a handle a worker took, unless the resource is already + /// closing — in which case the handle is dropped to release the OS + /// resource rather than re-inserted into a closing resource. + fn restore_handle(&self, handle: IoHandle) { + if self.closed.load(Ordering::SeqCst) { + let _ = close_io_handle(handle); + return; + } + *self + .handle + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(handle); + } +} + +impl HostResource for IoResource { + fn begin_close(&mut self, _reason: ResourceCloseReason) -> ResourceResult { + self.closed.store(true, Ordering::SeqCst); + if let Some(handle) = self.take_handle() { + close_io_handle(handle).map_err(|error| { + ResourceError::new( + ResourceErrorCode::ResourceCleanupFailed, + "io::resource", + error.to_string(), + ) + })?; + } + Ok(CloseProgress::Ready) + } +} + +/// Shared state between one IO worker thread, its [`IoOpDriver`] operation, +/// and [`poll_builtin_io_op`] on the VM thread. +/// +/// The worker writes the terminal [`signal`](IoOpShared::signal), the +/// guest-visible [`value`](IoOpShared::value), and any opened handle or +/// close target; the driver reflects the signal into the operation registry +/// and the VM wrapper reads the value out of the mailbox after the registry +/// drive returns terminal. +struct IoOpShared { + cancelled: AtomicBool, + worker_done: AtomicBool, + signal: Mutex>>, + value: Mutex>>, + opened: Mutex>, + target: Mutex>, + waker: Mutex>, + quiescence_waker: Mutex>, + worker: Mutex>>, + cancel_hook: Mutex>>, +} + +impl IoOpShared { + fn new() -> Self { + Self { + cancelled: AtomicBool::new(false), + worker_done: AtomicBool::new(false), + signal: Mutex::new(None), + value: Mutex::new(None), + opened: Mutex::new(None), + target: Mutex::new(None), + waker: Mutex::new(None), + quiescence_waker: Mutex::new(None), + worker: Mutex::new(None), + cancel_hook: Mutex::new(None), + } + } + + fn mark_worker_done(&self) { + self.worker_done.store(true, Ordering::Release); + if let Some(waker) = self + .quiescence_waker + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) + .take() + { + waker.wake(); + } + } + + fn is_quiescent(&self) -> bool { + self.worker_done.load(Ordering::Acquire) + } + + fn register_quiescence_waker(&self, waker: &Waker) { + let mut guard = self + .quiescence_waker + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()); + if self.is_quiescent() { + return; + } + *guard = Some(waker.clone()); + if self.is_quiescent() + && let Some(waker) = guard.take() + { + waker.wake(); + } + } + + fn install_cancel_hook(&self, hook: impl FnOnce() + Send + 'static) { + let mut hook = Some(Box::new(hook) as Box); + { + let mut guard = self + .cancel_hook + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()); + if !self.cancelled.load(Ordering::Acquire) { + *guard = hook.take(); + } + } + if let Some(hook) = hook { + hook(); + } + } + + fn cancel_work(&self) { + if let Some(hook) = self + .cancel_hook + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) + .take() + { + hook(); + } + } + + fn set_worker(&self, worker: JoinHandle<()>) { + *self + .worker + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(worker); + } + + fn join_worker(&self) -> bool { + let worker = self + .worker + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) + .take(); + worker.is_some_and(|worker| worker.join().is_err()) + } + + /// The worker's terminal publish: stores the signal and wakes any + /// registered waker (check-register-double-check in the driver's poll). + fn publish(&self, signal: Result<(), String>) { + *self + .signal + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(signal); + if let Some(waker) = self + .waker + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) + .take() + { + waker.wake(); + } + } + + fn take_signal(&self) -> Option> { + self.signal + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) + .take() + } + + fn register_waker(&self, waker: &Waker) { + *self + .waker + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(waker.clone()); + } + + /// The worker's failure path: records the guest-visible `VmError` in the + /// value mailbox and publishes a textual signal for the operation driver. + fn fail(&self, error: VmError) { + let message = error.to_string(); + *self + .value + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(Err(error)); + self.publish(Err(message)); + } + + /// The worker's success path: records the guest-visible value and + /// publishes a success signal. + fn succeed(&self, value: CallReturn) { + *self + .value + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(Ok(value)); + self.publish(Ok(())); + } +} + +/// A concrete [`HostOperation`] driver for one pending IO operation. +/// +/// The worker thread performs the actual IO; this driver reflects the +/// worker's terminal signal into the operation registry and honours +/// cancellation by flagging the shared state so the worker aborts promptly. +struct IoOpDriver { + shared: Arc, + name: String, +} + +impl IoOpDriver { + fn new(shared: Arc, name: impl Into) -> Self { + Self { + shared, + name: name.into(), + } + } + + fn worker_failed(&self, message: impl Into) -> Poll> { + Poll::Ready(Err(OperationError::new( + OperationErrorCode::OperationDriverFailed, + "io::operation", + message, + ))) + } +} + +impl HostOperation for IoOpDriver { + fn poll(&mut self, cx: &mut Context<'_>) -> Poll> { + if !self.shared.is_quiescent() { + self.shared.register_waker(cx.waker()); + self.shared.register_quiescence_waker(cx.waker()); + if !self.shared.is_quiescent() { + return Poll::Pending; + } + } + if self.shared.cancelled.load(Ordering::Acquire) { + return self.worker_failed(format!("{} was cancelled", self.name)); + } + match self.shared.take_signal() { + Some(Ok(())) => Poll::Ready(Ok(())), + Some(Err(message)) => self.worker_failed(message), + None => self.worker_failed(format!( + "{} worker terminated without a completion signal", + self.name + )), + } + } + + fn cancel(&mut self, _reason: OperationCancelReason) -> OperationResult<()> { + self.shared.cancelled.store(true, Ordering::Release); + self.shared.cancel_work(); + Ok(()) + } + + fn is_quiescent(&self) -> bool { + self.shared.is_quiescent() + } + + fn register_quiescence_waker(&mut self, cx: &Context<'_>) { + self.shared.register_quiescence_waker(cx.waker()); + } + + fn cancel_and_wait(&mut self, reason: OperationCancelReason) -> OperationResult<()> { + self.cancel(reason)?; + if self.shared.join_worker() { + return Err(OperationError::new( + OperationErrorCode::OperationDriverFailed, + "io::operation", + format!("{} worker panicked while cancelling", self.name), + )); + } + Ok(()) + } +} + +impl Drop for IoOpDriver { + fn drop(&mut self) { + if !self.shared.is_quiescent() { + self.shared.cancelled.store(true, Ordering::Release); + self.shared.cancel_work(); + } + let _ = self.shared.join_worker(); + } +} + +/// Cancels one pending builtin IO operation through the execution scope. +pub(crate) fn cancel_pending_op(vm: &mut Vm, op_id: HostOpId) { + let Ok(id) = OperationId::from_raw(op_id) else { + return; + }; + // Drop the completion mailbox from the adapter-declared scope state; the + // operation's driver is cancelled through the registry (which forwards to + // the driver's `cancel`). + if let Ok(state) = io_mailbox(vm) { + state.pending_results.remove(&op_id); + } + let _ = vm + .execution_scope() + .cancel_operation(id, OperationCancelReason::Requested); +} + +/// Polls one pending builtin IO operation through the execution scope's +/// operation registry, delivering the worker's guest-visible value. +pub(crate) fn poll_builtin_io_op( + vm: &mut Vm, + op_id: HostOpId, + cx: &mut Context<'_>, +) -> Poll> { + let id = match OperationId::from_raw(op_id) { + Ok(id) => id, + Err(error) => { + return Poll::Ready(Err(VmError::HostError(format!( + "invalid builtin io op {op_id}: {error}" + )))); + } + }; + + let poll_result = vm.execution_scope().poll_operation(id, cx); + match poll_result { + Poll::Pending => Poll::Pending, + Poll::Ready(Err(error)) => { + // Drop the completion mailbox from the adapter-declared scope + // state; the operation itself already failed terminal. + if let Ok(state) = io_mailbox(vm) { + state.pending_results.remove(&op_id); + } + Poll::Ready(Err(VmError::HostError(format!( + "builtin io op {op_id} failed: {error}" + )))) + } + Poll::Ready(Ok(outcome)) => { + // The worker wrote the authoritative guest-visible result into + // the completion mailbox before signalling terminal. Extract the + // mailbox entry (an `Arc`) so the scope borrow ends before the + // resource insertion below re-borrows the scope. + let shared = match io_mailbox(vm) { + Ok(state) => { + let Some(shared) = state.pending_results.remove(&op_id) else { + return Poll::Ready(Err(VmError::HostError(format!( + "builtin io op {op_id} has no completion mailbox" + )))); + }; + shared + } + Err(error) => { + return Poll::Ready(Err(VmError::HostError(format!( + "builtin io op {op_id} mailbox unavailable: {error}" + )))); + } + }; + if matches!( + outcome, + crate::vm::operation::driver::OperationOutcome::Cancelled(_) + ) || shared.cancelled.load(Ordering::Acquire) + { + return Poll::Ready(Err(VmError::HostError( + "IO operation cancelled".to_string(), + ))); + } + + // An opened handle (io::open / io::popen) becomes a typed IO + // resource in the scope; the script-visible handle is its raw + // resource token. + if let Some(handle) = shared + .opened + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) + .take() + { + let token = match vm.execution_scope().push_resource(IoResource::new(handle)) { + Ok(token) => token, + Err(error) => { + return Poll::Ready(Err(VmError::HostError(format!( + "builtin io op {op_id} resource insert failed: {error}" + )))); + } + }; + *shared + .value + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) = + Some(Ok(CallReturn::one(Value::Int(token.handle().raw() as i64)))); + } + + // A closed handle (io::close) retires the exact resource entry + // through the generic scope close (exact-once). + if let Some(target) = shared + .target + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) + .take() + { + let _ = vm + .execution_scope() + .close_resource::(target, ResourceCloseReason::Requested); + } + + let value = shared + .value + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) + .take(); + match value { + Some(value) => Poll::Ready(value), + None => Poll::Ready(Err(VmError::HostError(format!( + "builtin io op {op_id} completed without a result" + )))), + } + } + } +} + +/// Returns the adapter-declared IO scope state (the per-op completion +/// mailbox), creating the empty default on first access while the scope is +/// Active. The state is owned by the execution-scope arena, so it is +/// destroyed with the scope on reset and recreated lazily on next use. +fn io_mailbox(vm: &mut Vm) -> VmResult<&mut IoState> { + vm.execution_scope() + .scope_state_or_insert_with(IoState::default) + .map_err(|error| VmError::HostError(format!("io scope state unavailable: {error}"))) +} + +/// Maximum UTF-8 byte length passed to `thread::Builder::name` for an IO +/// worker. The sanitized ASCII name also avoids embedded NULs and platform +/// surprises from an operation name supplied by a future caller. +const IO_WORKER_THREAD_NAME_MAX_LEN: usize = 32; + +fn io_worker_thread_name(operation: &str) -> String { + let mut name = String::from("pd-vm-io-"); + for byte in operation.bytes() { + if name.len() == IO_WORKER_THREAD_NAME_MAX_LEN { + break; + } + let safe = match byte { + b'a'..=b'z' | b'A'..=b'Z' | b'0'..=b'9' | b'_' | b'-' => byte, + _ => b'_', + }; + name.push(safe as char); + } + name +} + +struct WorkerCompletion { + shared: Arc, +} + +impl Drop for WorkerCompletion { + fn drop(&mut self) { + self.shared.mark_worker_done(); + } +} + +/// Spawns a worker thread for an IO operation and registers its driver in +/// the VM's execution scope. Returns the packed [`OperationId`] raw value +/// to hand to the guest as the pending op id. +fn schedule_io_task( + vm: &mut Vm, + name: &str, + work: impl FnOnce(&IoOpShared) + Send + 'static, +) -> VmResult { + let name = name.to_string(); + let shared = Arc::new(IoOpShared::new()); + let driver_shared = Arc::clone(&shared); + let worker_shared = Arc::clone(&shared); + let worker_name = name.clone(); + + let op_id = vm + .execution_scope() + .start_operation(OperationSpec::new(IoOpDriver::new(driver_shared, name))) + .map_err(|error| { + VmError::HostError(format!( + "failed to start io operation '{}': {error}", + worker_name + )) + })?; + let raw = op_id.raw(); + let thread_name = io_worker_thread_name(&worker_name); + + std::thread::Builder::new() + .name(thread_name) + .spawn(move || { + let _completion = WorkerCompletion { + shared: Arc::clone(&worker_shared), + }; + if worker_shared.cancelled.load(Ordering::Acquire) { + worker_shared.publish(Err(format!("io operation '{worker_name}' was cancelled"))); + return; + } + work(&worker_shared); + }) + .map(|worker| { + shared.set_worker(worker); + }) + .map_err(|error| { + // Roll back the registered operation so no orphaned op lingers. + shared.mark_worker_done(); + let _ = vm + .execution_scope() + .abort_operation(op_id, OperationCancelReason::Requested); + VmError::HostError(format!("failed to spawn io task: {error}")) + })?; + + io_mailbox(vm)?.pending_results.insert(raw, shared); + Ok(raw) +} + +/// Opens a file handle for runtime I/O. +#[pd_host_function(name = "io::open")] +pub(super) fn builtin_io_open( + vm: &mut Vm, + path: &str, + mode: &str, +) -> VmResult> { + let writes = matches!(mode, "w" | "a" | "r+" | "w+" | "a+"); + let path = authorize_blocking_io_path(vm, path, writes)? + .display() + .to_string(); + let mode = mode.to_string(); + let op_id = schedule_io_task(vm, "io::open", move |shared| { + let mut options = OpenOptions::new(); + match mode.as_str() { + "r" => { + options.read(true); + } + "w" => { + options.write(true).create(true).truncate(true); + } + "a" => { + options.write(true).create(true).append(true); + } + "r+" => { + options.read(true).write(true); + } + "w+" => { + options.read(true).write(true).create(true).truncate(true); + } + "a+" => { + options.read(true).write(true).create(true).append(true); + } + other => { + shared.fail(VmError::HostError(format!( + "unsupported io_open mode '{other}', expected r/w/a/r+/w+/a+", + ))); + return; + } + } + + match options.open(path) { + Ok(file) => { + *shared + .opened + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(IoHandle::File(file)); + shared.publish(Ok(())); + } + Err(err) => { + shared.fail(VmError::HostError(format!("io_open failed: {err}"))); + } + } + })?; + Ok(HostCallResult::Pending(op_id)) +} + +/// Starts a child process and returns a process-backed handle. +#[pd_host_function(name = "io::popen")] +pub(super) fn builtin_io_popen( + vm: &mut Vm, + command: &str, + mode: &str, +) -> VmResult> { + if mode != "r" && mode != "w" { + return Err(VmError::HostError(format!( + "unsupported io_popen mode '{mode}', expected r or w" + ))); + } + if super::io_policy(vm) + .as_ref() + .is_some_and(|policy| !policy.allow_process) + { + return Err(VmError::HostError( + "io_popen requires the process capability".to_string(), + )); + } + let command = command.to_string(); + let mode = mode.to_string(); + let op_id = schedule_io_task(vm, "io::popen", move |shared| { + let child = match spawn_shell_command(command.as_str(), mode.as_str()) { + Ok(child) => child, + Err(err) => { + shared.fail(err); + return; + } + }; + let child_pid = child.id(); + shared.install_cancel_hook(move || terminate_process_tree(child_pid)); + let handle = match mode.as_str() { + "r" => { + if child.stdout.is_none() { + let err = + VmError::HostError("io_popen('r') did not provide stdout pipe".to_string()); + shared.fail(err); + return; + } + IoHandle::PopenRead { child } + } + "w" => { + if child.stdin.is_none() { + let err = + VmError::HostError("io_popen('w') did not provide stdin pipe".to_string()); + shared.fail(err); + return; + } + IoHandle::PopenWrite { child } + } + _ => unreachable!("mode validated above"), + }; + *shared + .opened + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(handle); + shared.publish(Ok(())); + })?; + Ok(HostCallResult::Pending(op_id)) +} + +/// Reads all remaining text from an I/O handle. +#[pd_host_function(name = "io::read_all")] +pub(super) fn builtin_io_read_all(vm: &mut Vm, handle_id: i64) -> VmResult> { + let (_handle, resource) = io_resource_for_handle(vm, handle_id)?; + let op_id = schedule_io_task(vm, "io::read_all", move |shared| { + let mut handle = match resource.take_handle() { + Some(handle) => handle, + None => { + let err = VmError::HostError("io_read_all handle is already closing".to_string()); + shared.fail(err); + return; + } + }; + install_process_cancel_hook(shared, &handle); + let mut out = String::new(); + let result = match &mut handle { + IoHandle::File(file) => file + .read_to_string(&mut out) + .map_err(|err| VmError::HostError(format!("io_read_all failed: {err}"))) + .map(|_| CallReturn::one(Value::string(out))), + IoHandle::PopenRead { child } => { + let stdout = match child.stdout.as_mut() { + Some(stdout) => stdout, + None => { + resource.restore_handle(handle); + let err = VmError::HostError( + "io_read_all popen handle missing stdout".to_string(), + ); + shared.fail(err); + return; + } + }; + stdout + .read_to_string(&mut out) + .map_err(|err| VmError::HostError(format!("io_read_all failed: {err}"))) + .map(|_| CallReturn::one(Value::string(out))) + } + IoHandle::PopenWrite { .. } => Err(VmError::HostError( + "io_read_all requires a readable handle".to_string(), + )), + }; + resource.restore_handle(handle); + match result { + Ok(value) => { + shared.succeed(value); + } + Err(err) => { + shared.fail(err); + } + } + })?; + Ok(HostCallResult::Pending(op_id)) +} + +/// Reads a single line of text from an I/O handle. +#[pd_host_function(name = "io::read_line")] +pub(super) fn builtin_io_read_line( + vm: &mut Vm, + handle_id: i64, +) -> VmResult> { + let (_handle, resource) = io_resource_for_handle(vm, handle_id)?; + let op_id = schedule_io_task(vm, "io::read_line", move |shared| { + let mut handle = match resource.take_handle() { + Some(handle) => handle, + None => { + let err = VmError::HostError("io_read_line handle is already closing".to_string()); + shared.fail(err); + return; + } + }; + install_process_cancel_hook(shared, &handle); + let result = match &mut handle { + IoHandle::File(file) => { + read_line_from_reader(file).map(|line| CallReturn::one(Value::string(line))) + } + IoHandle::PopenRead { child } => { + let stdout = match child.stdout.as_mut() { + Some(stdout) => stdout, + None => { + resource.restore_handle(handle); + let err = VmError::HostError( + "io_read_line popen handle missing stdout".to_string(), + ); + shared.fail(err); + return; + } + }; + read_line_from_reader(stdout).map(|line| CallReturn::one(Value::string(line))) + } + IoHandle::PopenWrite { .. } => Err(VmError::HostError( + "io_read_line requires a readable handle".to_string(), + )), + }; + resource.restore_handle(handle); + match result { + Ok(value) => { + shared.succeed(value); + } + Err(err) => { + shared.fail(err); + } + } + })?; + Ok(HostCallResult::Pending(op_id)) +} + +/// Writes text to an I/O handle. +#[pd_host_function(name = "io::write")] +pub(super) fn builtin_io_write( + vm: &mut Vm, + handle_id: i64, + text: &str, +) -> VmResult> { + if super::io_policy(vm) + .as_ref() + .is_some_and(|policy| text.len() > policy.max_write_bytes) + { + return Err(VmError::HostError( + "io_write exceeded write limit".to_string(), + )); + } + let bytes = text.as_bytes().to_vec(); + let (_handle, resource) = io_resource_for_handle(vm, handle_id)?; + let op_id = schedule_io_task(vm, "io::write", move |shared| { + let mut handle = match resource.take_handle() { + Some(handle) => handle, + None => { + let err = VmError::HostError("io_write handle is already closing".to_string()); + shared.fail(err); + return; + } + }; + install_process_cancel_hook(shared, &handle); + let result = match &mut handle { + IoHandle::File(file) => file + .write(&bytes) + .map_err(|err| VmError::HostError(format!("io_write failed: {err}"))) + .map(|written| CallReturn::one(Value::Int(written as i64))), + IoHandle::PopenWrite { child } => { + let stdin = match child.stdin.as_mut() { + Some(stdin) => stdin, + None => { + resource.restore_handle(handle); + let err = + VmError::HostError("io_write popen handle missing stdin".to_string()); + shared.fail(err); + return; + } + }; + stdin + .write(&bytes) + .map_err(|err| VmError::HostError(format!("io_write failed: {err}"))) + .map(|written| CallReturn::one(Value::Int(written as i64))) + } + IoHandle::PopenRead { .. } => Err(VmError::HostError( + "io_write requires a writable handle".to_string(), + )), + }; + resource.restore_handle(handle); + match result { + Ok(value) => { + shared.succeed(value); + } + Err(err) => { + shared.fail(err); + } + } + })?; + Ok(HostCallResult::Pending(op_id)) +} + +/// Flushes buffered output for an I/O handle. +#[pd_host_function(name = "io::flush")] +pub(super) fn builtin_io_flush(vm: &mut Vm, handle_id: i64) -> VmResult> { + let (_handle, resource) = io_resource_for_handle(vm, handle_id)?; + let op_id = schedule_io_task(vm, "io::flush", move |shared| { + let mut handle = match resource.take_handle() { + Some(handle) => handle, + None => { + let err = VmError::HostError("io_flush handle is already closing".to_string()); + shared.fail(err); + return; + } + }; + install_process_cancel_hook(shared, &handle); + let result = match &mut handle { + IoHandle::File(file) => file + .flush() + .map_err(|err| VmError::HostError(format!("io_flush failed: {err}"))) + .map(|_| CallReturn::one(Value::Bool(true))), + IoHandle::PopenWrite { child } => { + let stdin = match child.stdin.as_mut() { + Some(stdin) => stdin, + None => { + resource.restore_handle(handle); + let err = + VmError::HostError("io_flush popen handle missing stdin".to_string()); + shared.fail(err); + return; + } + }; + stdin + .flush() + .map_err(|err| VmError::HostError(format!("io_flush failed: {err}"))) + .map(|_| CallReturn::one(Value::Bool(true))) + } + IoHandle::PopenRead { .. } => Ok(CallReturn::one(Value::Bool(true))), + }; + resource.restore_handle(handle); + match result { + Ok(value) => { + shared.succeed(value); + } + Err(err) => { + shared.fail(err); + } + } + })?; + Ok(HostCallResult::Pending(op_id)) +} + +/// Closes an I/O handle. +#[pd_host_function(name = "io::close")] +pub(super) fn builtin_io_close(vm: &mut Vm, handle_id: i64) -> VmResult> { + let (target, resource) = io_resource_for_handle(vm, handle_id)?; + let op_id = schedule_io_task(vm, "io::close", move |shared| { + // Close the underlying handle exactly once on the worker thread. + let result = match resource.take_handle() { + Some(handle) => { + install_process_cancel_hook(shared, &handle); + close_io_handle(handle) + } + None => Err(VmError::HostError( + "io_close handle is already closing".to_string(), + )), + }; + *shared + .target + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(target); + match result { + Ok(()) => { + shared.succeed(CallReturn::one(Value::Bool(true))); + } + Err(err) => { + shared.fail(err); + } + } + })?; + Ok(HostCallResult::Pending(op_id)) +} + +/// Returns whether a file system path exists. +#[pd_host_function(name = "io::exists")] +pub(super) fn builtin_io_exists(vm: &mut Vm, path: &str) -> VmResult> { + let path = authorize_blocking_io_path(vm, path, false)? + .display() + .to_string(); + let op_id = schedule_io_task(vm, "io::exists", move |shared| { + shared.succeed(CallReturn::one(Value::Bool( + std::path::Path::new(path.as_str()).exists(), + ))); + })?; + Ok(HostCallResult::Pending(op_id)) +} + +fn install_process_cancel_hook(shared: &IoOpShared, handle: &IoHandle) { + let pid = match handle { + IoHandle::PopenRead { child } | IoHandle::PopenWrite { child } => child.id(), + IoHandle::File(_) => return, + }; + shared.install_cancel_hook(move || terminate_process_tree(pid)); +} + +fn terminate_process_tree(pid: u32) { + #[cfg(unix)] + { + let Ok(pid) = libc::pid_t::try_from(pid) else { + return; + }; + // `spawn_shell_command` puts the shell in its own process group, so a + // negative pid terminates the shell and descendants without touching + // the VM process group. + unsafe { + libc::kill(-pid, libc::SIGKILL); + } + } + #[cfg(windows)] + { + let _ = Command::new("taskkill") + .args(["/T", "/F", "/PID", &pid.to_string()]) + .status(); + } + #[cfg(not(any(unix, windows)))] + { + let _ = pid; + } +} + +fn spawn_shell_command(command: &str, mode: &str) -> VmResult { + let mut process = if cfg!(windows) { + let mut cmd = Command::new("cmd"); + cmd.arg("/C").arg(command); + cmd + } else { + let mut cmd = Command::new("sh"); + cmd.arg("-c").arg(command); + cmd + }; + + #[cfg(unix)] + { + use std::os::unix::process::CommandExt as _; + process.process_group(0); + } + + match mode { + "r" => { + process.stdout(Stdio::piped()).stdin(Stdio::null()); + } + "w" => { + process.stdin(Stdio::piped()).stdout(Stdio::null()); + } + _ => {} + } + + process + .spawn() + .map_err(|err| VmError::HostError(format!("io_popen failed: {err}"))) +} + +/// Parses a script-visible integer handle into a typed scope token and +/// returns the raw scope handle plus shared resource cells, validating +/// staleness and type through the generic typed table. +fn io_resource_for_handle( + vm: &mut Vm, + handle_id: i64, +) -> VmResult<(ResourceHandle, Arc)> { + let handle = io_parse_handle(handle_id)?; + let token = vm + .execution_scope() + .resources() + .typed::(handle) + .map_err(|error| { + VmError::HostError(format!( + "io handle {handle_id} is not a live IO handle: {error}" + )) + })?; + let resource = vm + .execution_scope() + .resources() + .get::(&token) + .map_err(|error| { + VmError::HostError(format!("io handle {handle_id} borrow failed: {error}")) + })?; + // Clone the shared cells so the worker can take/restore the handle while + // the resource itself stays in the scope table. + Ok(( + handle, + Arc::new(IoResource { + handle: Arc::clone(&resource.handle), + closed: Arc::clone(&resource.closed), + }), + )) +} + +fn io_parse_handle(handle_id: i64) -> VmResult { + if handle_id <= 0 { + return Err(VmError::HostError(format!( + "invalid io handle id {handle_id}; expected positive handle id" + ))); + } + ResourceHandle::from_raw(handle_id as u64) + .map_err(|error| VmError::HostError(format!("invalid io handle id {handle_id}: {error}"))) +} + +/// Authorizes one IO path against the configured policy, mirroring the +/// async path: a policy with no matching allowed root denies the path. +fn authorize_blocking_io_path(vm: &Vm, path: &str, writes: bool) -> VmResult { + let requested = PathBuf::from(path); + let Some(policy) = super::io_policy(vm) else { + return Ok(requested); + }; + if writes && !policy.allow_write { + return Err(VmError::HostError( + "io path write requires the write capability".to_string(), + )); + } + let absolute = if requested.is_absolute() { + requested + } else { + std::env::current_dir() + .map_err(|error| VmError::HostError(format!("io path resolution failed: {error}")))? + .join(requested) + }; + let canonical = canonicalize_blocking_target(&absolute)?; + for root in &policy.allowed_roots { + let root = std::fs::canonicalize(Path::new(root)).map_err(|error| { + VmError::HostError(format!( + "io allowed root '{root}' cannot be resolved: {error}" + )) + })?; + if canonical.starts_with(root) { + return Ok(canonical); + } + } + Err(VmError::HostError(format!( + "io path '{}' is outside the allowed roots", + canonical.display() + ))) +} + +fn canonicalize_blocking_target(path: &Path) -> VmResult { + if path.exists() { + return std::fs::canonicalize(path) + .map_err(|error| VmError::HostError(format!("io path resolution failed: {error}"))); + } + let parent = path.parent().unwrap_or_else(|| Path::new(".")); + let canonical_parent = std::fs::canonicalize(parent) + .map_err(|error| VmError::HostError(format!("io path resolution failed: {error}")))?; + let name = path + .file_name() + .ok_or_else(|| VmError::HostError("io path has no file name".to_string()))?; + Ok(canonical_parent.join(name)) +} + +fn close_io_handle(mut handle: IoHandle) -> VmResult<()> { + match &mut handle { + IoHandle::File(file) => { + file.flush().ok(); + } + IoHandle::PopenRead { child } => { + child + .wait() + .map_err(|err| VmError::HostError(format!("io_close popen wait failed: {err}")))?; + } + IoHandle::PopenWrite { child } => { + let _ = child.stdin.take(); + child + .wait() + .map_err(|err| VmError::HostError(format!("io_close popen wait failed: {err}")))?; + } + } + Ok(()) +} + +fn read_line_from_reader(reader: &mut impl Read) -> VmResult { + let mut bytes = Vec::new(); + let mut one = [0u8; 1]; + loop { + let read = reader + .read(&mut one) + .map_err(|err| VmError::HostError(format!("io_read_line failed: {err}")))?; + if read == 0 { + break; + } + bytes.push(one[0]); + if one[0] == b'\n' { + break; + } + } + Ok(String::from_utf8_lossy(&bytes).into_owned()) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn io_worker_thread_name_is_sanitized_and_bounded() { + assert_eq!( + io_worker_thread_name("io::read_all"), + "pd-vm-io-io__read_all" + ); + let name = io_worker_thread_name("io::operation/with spaces\0 and a very long suffix"); + assert!(name.len() <= IO_WORKER_THREAD_NAME_MAX_LEN); + assert!(name.is_ascii()); + assert!( + name.bytes() + .all(|byte| byte.is_ascii_alphanumeric() || byte == b'_' || byte == b'-') + ); + assert!(!name.contains('\0')); + } + + #[test] + fn io_driver_waits_for_worker_completion_before_reporting_ready() { + let shared = Arc::new(IoOpShared::new()); + let release = Arc::new(AtomicBool::new(false)); + let worker_shared = Arc::clone(&shared); + let worker_release = Arc::clone(&release); + let worker = std::thread::spawn(move || { + while !worker_release.load(Ordering::Acquire) { + std::thread::yield_now(); + } + worker_shared.publish(Ok(())); + worker_shared.mark_worker_done(); + }); + shared.set_worker(worker); + let mut driver = IoOpDriver::new(Arc::clone(&shared), "io::test"); + let mut cx = Context::from_waker(Waker::noop()); + + assert!(matches!(driver.poll(&mut cx), Poll::Pending)); + assert!(!driver.is_quiescent()); + + release.store(true, Ordering::Release); + while !driver.is_quiescent() { + std::thread::yield_now(); + } + assert!(matches!(driver.poll(&mut cx), Poll::Ready(Ok(())))); + } +} diff --git a/src/builtins/runtime/io/mod.rs b/src/builtins/runtime/io/mod.rs new file mode 100644 index 00000000..8074a495 --- /dev/null +++ b/src/builtins/runtime/io/mod.rs @@ -0,0 +1,71 @@ +//! IO builtin host implementation. +//! +//! At this layer IO is blocking-only: it drives IO through worker threads +//! registered as concrete [`HostOperation`] drivers in the execution scope. +//! Live handles are [`IoResource`]s owned by the VM's execution scope and +//! in-flight IO work is driven by concrete operation drivers registered in +//! the same scope. +//! +//! The capability system (restricted registries and explicit grants) is +//! introduced by the public host SDK layer; before that layer exists, +//! [`io_policy`] returns only the configured persistent [`IoPolicy`] held in +//! the generic module-state store. + +use super::borrow_arg; +use crate::vm::Vm; + +pub(super) use super::HostCallResult; + +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct IoPolicy { + pub allowed_roots: Vec, + pub allow_write: bool, + pub allow_process: bool, + pub max_read_bytes: usize, + pub max_write_bytes: usize, +} + +impl Default for IoPolicy { + fn default() -> Self { + Self { + allowed_roots: Vec::new(), + allow_write: false, + allow_process: false, + max_read_bytes: 1024 * 1024, + max_write_bytes: 1024 * 1024, + } + } +} + +/// I/O host configuration owned by the I/O host implementation. +pub trait IoHostExt { + fn configure_io(&mut self, policy: IoPolicy); + fn clear_io_configuration(&mut self); +} + +impl IoHostExt for Vm { + fn configure_io(&mut self, mut policy: IoPolicy) { + policy.allowed_roots.sort(); + policy.allowed_roots.dedup(); + // Adapter-declared policy stored in the generic module-state store: + // module-level policy survives execution-scope reset (an embedder's + // roots remain in force across `reset_for_reuse`), while the adapter's + // per-invocation runtime state lives in the scope arena. + self.host.set_module_state(policy); + } + + fn clear_io_configuration(&mut self) { + self.host.remove_module_state::(); + } +} + +pub(super) fn io_policy(vm: &Vm) -> Option { + vm.host.get_module_state::().cloned() +} + +#[cfg(target_arch = "wasm32")] +pub(super) use super::io_wasm::*; +#[cfg(not(target_arch = "wasm32"))] +mod blocking; +#[cfg(not(target_arch = "wasm32"))] +pub(crate) use blocking::*; diff --git a/src/builtins/runtime/io_wasm.rs b/src/builtins/runtime/io_wasm.rs index 3b2998e2..651daad6 100644 --- a/src/builtins/runtime/io_wasm.rs +++ b/src/builtins/runtime/io_wasm.rs @@ -3,15 +3,13 @@ use std::task::{Context, Poll}; use pd_host_function::pd_host_function; use super::HostCallResult; -use crate::vm::{CallReturn, HostOpId, Value, Vm, VmError, VmResult}; +use crate::vm::{CallReturn, HostOpId, Vm, VmError, VmResult}; -pub(crate) struct IoState; - -impl Default for IoState { - fn default() -> Self { - Self - } -} +/// There are no pending native I/O workers on wasm32. The generic VM +/// cancellation hook still calls into the selected I/O backend, so keep the +/// wasm implementation a deliberate no-op with the same feature-neutral +/// signature as the native backends. +pub(super) fn cancel_pending_op(_vm: &mut Vm, _op_id: HostOpId) {} pub(super) fn poll_builtin_io_op( _vm: &mut Vm, @@ -23,8 +21,6 @@ pub(super) fn poll_builtin_io_op( )))) } -pub(super) fn close_all_handles(_vm: &mut Vm) {} - /// Opens a file handle for runtime I/O. #[pd_host_function(name = "io::open")] pub(super) fn builtin_io_open( diff --git a/src/builtins/runtime/mod.rs b/src/builtins/runtime/mod.rs index 67fac4fc..715a3c86 100644 --- a/src/builtins/runtime/mod.rs +++ b/src/builtins/runtime/mod.rs @@ -3,7 +3,8 @@ use std::task::{Context, Poll}; use crate::builtins::BuiltinFunction; -use crate::vm::{CallOutcome, CallReturn, HostOpId, Value, Vm, VmResult}; +#[allow(unused_imports)] +use crate::vm::{CallOutcome, CallReturn, HostOpId, Value, Vm, VmError, VmResult}; mod aot; mod bytes; @@ -19,12 +20,15 @@ mod map_iter; mod math; pub(crate) mod print; pub(crate) mod regex; +#[cfg(all(feature = "sqlite", not(target_arch = "wasm32")))] +pub(crate) mod sqlite; mod typed; #[cfg(target_arch = "wasm32")] use io_wasm as io; -pub(crate) use io::IoState; +#[cfg(not(target_arch = "wasm32"))] +pub use io::{IoHostExt, IoPolicy}; pub use typed::HostCallResult; use typed::{ AnyValue, IntoBuiltinCallOutcome, IntoHostCallOutcome, NumberValue, UnknownValue, VmArray, @@ -136,8 +140,35 @@ pub(crate) fn poll_builtin_io_op( io::poll_builtin_io_op(vm, op_id, cx) } -pub(crate) fn close_all_handles(vm: &mut Vm) { - io::close_all_handles(vm); +/// Cancels one pending SQLite operation. The generic VM (feature-neutral) +/// calls this hook unconditionally; on builds without the SQLite adapter the +/// delegation below is compiled out and the call is a no-op. +pub(crate) fn cancel_builtin_sqlite_op(vm: &mut Vm, op_id: HostOpId) { + #[cfg(all(feature = "sqlite", not(target_arch = "wasm32")))] + sqlite::cancel_pending_op(vm, op_id); + #[cfg(not(all(feature = "sqlite", not(target_arch = "wasm32"))))] + let _ = (vm, op_id); +} + +/// Polls one pending SQLite operation. The generic VM (feature-neutral) calls +/// this hook unconditionally; on builds without the SQLite adapter it reports +/// an unsupported-operations error. +pub(crate) fn poll_builtin_sqlite_op( + vm: &mut Vm, + op_id: HostOpId, + cx: &mut Context<'_>, +) -> Poll> { + #[cfg(all(feature = "sqlite", not(target_arch = "wasm32")))] + { + sqlite::poll_pending_op(vm, op_id, cx) + } + #[cfg(not(all(feature = "sqlite", not(target_arch = "wasm32"))))] + { + let _ = (vm, cx); + Poll::Ready(Err(VmError::HostError(format!( + "builtin sqlite op {op_id} is unsupported in this build" + )))) + } } #[cfg(test)] diff --git a/src/builtins/runtime/namespaces.rs b/src/builtins/runtime/namespaces.rs index 62638c1c..b8dd31da 100644 --- a/src/builtins/runtime/namespaces.rs +++ b/src/builtins/runtime/namespaces.rs @@ -5,4 +5,5 @@ builtin_namespaces![ builtin_namespace!("json", "json", "JSON builtin namespace.", true), builtin_namespace!("jit", "jit", "JIT control builtin namespace.", true), builtin_namespace!("math", "math", "Numeric math builtin namespace.", true), + builtin_namespace!("sqlite", "sqlite", "SQLite database builtin namespace.", false), ]; diff --git a/src/builtins/runtime/sqlite.rs b/src/builtins/runtime/sqlite.rs new file mode 100644 index 00000000..0186c8e1 --- /dev/null +++ b/src/builtins/runtime/sqlite.rs @@ -0,0 +1,1709 @@ +//! Scoped SQLite host functions (optional `sqlite` feature). +//! +//! SQLite connections are typed [`HostResource`]s owned by the VM's +//! [`ExecutionScope`](crate::vm::execution_scope::ExecutionScope), exactly +//! like IO handles. Pending `sqlite::execute` / `sqlite::query` / +//! `sqlite::transaction` work is driven by concrete [`HostOperation`] +//! drivers registered in the same scope and polled/cancelled directly by the +//! operation registry. There is no poller table, no operation-owner enum, and +//! no callback-payload resource: the driver holds the shared connection slot +//! and the scope drives its lifecycle. +//! +//! Connection cleanup is adapter-owned: closing the resource (via +//! `sqlite::close`, VM reset, or scope drop) interrupts the connection +//! through the slot the resource owns and marks it closed. Pending drivers on +//! that connection observe the closed state and are retired through the +//! generic scope close, so no `close_resources_by_type` / +//! `cancel_operations_by_owner` helper is needed. +//! +//! Bounds preserved from the PR16 source: statement byte length, parameter +//! count and byte length, result rows/columns/bytes, connection count, +//! transaction statement count, and transaction deadline, plus SQL-safety +//! rejection and read-only enforcement. + +use std::collections::HashMap; +use std::fs; +use std::path::{Component, Path, PathBuf}; +use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; +use std::sync::{Arc, Mutex}; +use std::task::{Context, Poll, Waker}; +use std::thread; +use std::thread::JoinHandle; +use std::time::{Duration, Instant}; + +use pd_host_function::pd_host_function; +use rusqlite::hooks::{AuthAction, AuthContext, Authorization}; +use rusqlite::limits::Limit; +use rusqlite::types::{Value as SqlValue, ValueRef}; +use rusqlite::{Connection, OpenFlags, TransactionBehavior, params_from_iter}; + +use super::typed::{VmArrayRef, VmMapRef}; +use super::{HostCallResult, VmMap}; +use crate::vm::operation::driver::HostOperation; +use crate::vm::operation::error::{OperationError, OperationErrorCode, OperationResult}; +use crate::vm::operation::reason::OperationCancelReason; +use crate::vm::operation::{OperationId, OperationSpec}; +use crate::vm::resource::close::{CloseProgress, HostResource}; +use crate::vm::resource::error::ResourceResult; +use crate::vm::resource::{ResourceCloseReason, ResourceHandle}; +use crate::vm::{CallReturn, HostOpId, Value, Vm, VmError, VmResult}; + +/// SQLite `progress_handler` step cadence used to surface cancellation while a +/// statement runs. +const SQLITE_PROGRESS_STEPS: i32 = 1_000; + +/// Bounded SQLite connection/query limits, mirroring the PR16 source surface. +#[derive(Clone, Copy, Debug)] +pub struct SqliteLimits { + pub max_connections: usize, + pub max_statements: usize, + pub max_rows: usize, + pub max_columns: usize, + pub max_result_bytes: usize, + pub max_statement_bytes: usize, + pub max_parameters: usize, + pub max_parameter_bytes: usize, + pub max_pending_operations: usize, + pub max_transaction_ms: u64, + pub busy_timeout_ms: u64, +} + +impl Default for SqliteLimits { + fn default() -> Self { + Self { + max_connections: 16, + max_statements: 128, + max_rows: 1_000, + max_columns: 128, + max_result_bytes: 4 * 1024 * 1024, + max_statement_bytes: 1024 * 1024, + max_parameters: 128, + max_parameter_bytes: 1024 * 1024, + max_pending_operations: 32, + max_transaction_ms: 5_000, + busy_timeout_ms: 5_000, + } + } +} + +/// Embedding policy for the SQLite namespace. +#[derive(Clone, Debug, Default)] +pub struct SqlitePolicy { + pub database_root: Option, + pub allow_unsafe_sql: bool, + pub limits: SqliteLimits, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +enum OpenMode { + Memory, + ReadOnly, + ReadWrite, + ReadWriteCreate, +} + +struct OpenOptions { + path: String, + mode: OpenMode, + root: Option, + limits: SqliteLimits, + allow_unsafe_sql: bool, +} + +/// Shared, adapter-owned per-connection state. +/// +/// The connection itself is a [`Mutex`] (SQLite connections are +/// not thread-safe), serialized by the `execution` mutex so at most one +/// worker uses the connection at a time. The slot records the currently +/// executing operation and every in-flight operation on this connection so +/// close can retire them without a type-dispatched helper. +struct ConnectionSlot { + connection: Mutex, + execution: Mutex<()>, + /// The operation currently executing on this connection, if any. + active_operation: Mutex>, + /// Every in-flight operation scheduled against this connection. + pending: Mutex>, + /// Workers that have been scheduled but whose completion guard has not + /// retired yet. This closes the publish/unregister tail window. + live_workers: AtomicUsize, + /// Waker for a resource close waiting for `pending` to become empty. + close_waker: Mutex>, + interrupt: Arc, + limits: SqliteLimits, + allow_unsafe_sql: bool, + closed: AtomicBool, +} + +impl ConnectionSlot { + fn register(&self, id: OperationId) { + self.pending.lock().expect("sqlite pending lock").push(id); + self.live_workers.fetch_add(1, Ordering::Release); + } + + fn unregister(&self, id: OperationId) { + let removed = { + let mut pending = self.pending.lock().expect("sqlite pending lock"); + let before = pending.len(); + pending.retain(|candidate| *candidate != id); + pending.len() != before + }; + if !removed { + return; + } + let workers = self.live_workers.fetch_sub(1, Ordering::AcqRel) - 1; + if self.pending_count() == 0 + && workers == 0 + && let Some(waker) = self + .close_waker + .lock() + .expect("sqlite close waker lock") + .take() + { + waker.wake(); + } + } + + fn register_close_waker(&self, waker: &Waker) { + if self.pending_count() == 0 && self.live_workers.load(Ordering::Acquire) == 0 { + return; + } + { + let mut close_waker = self.close_waker.lock().expect("sqlite close waker lock"); + *close_waker = Some(waker.clone()); + } + if self.pending_count() == 0 + && self.live_workers.load(Ordering::Acquire) == 0 + && let Some(waker) = self + .close_waker + .lock() + .expect("sqlite close waker lock") + .take() + { + waker.wake(); + } + } + + fn drained(&self) -> bool { + self.pending_count() == 0 && self.live_workers.load(Ordering::Acquire) == 0 + } + + fn pending_count(&self) -> usize { + self.pending.lock().expect("sqlite pending lock").len() + } +} + +/// The typed connection resource stored in the execution scope. +/// +/// The slot is `Arc`-shared with worker threads so a closing resource does not +/// free the connection out from under an in-flight worker; the last Arc drops +/// the `Connection`. `begin_close` is exact-once: it marks the slot closed and +/// interrupts any currently executing statement so cancellation is prompt. +struct SqliteResource { + slot: Arc, + /// Adapter-owned live-connection counter (decremented on close). + open_connections: Arc, + counter_released: bool, +} + +impl SqliteResource { + fn new(slot: Arc, open_connections: Arc) -> Self { + Self { + slot, + open_connections, + counter_released: false, + } + } + + fn release_connection(&mut self) { + if !self.counter_released { + self.open_connections.fetch_sub(1, Ordering::SeqCst); + self.counter_released = true; + } + } +} + +impl HostResource for SqliteResource { + fn begin_close(&mut self, _reason: ResourceCloseReason) -> ResourceResult { + if !self.slot.closed.swap(true, Ordering::AcqRel) { + self.slot.interrupt.interrupt(); + } + if self.slot.drained() { + self.release_connection(); + Ok(CloseProgress::Ready) + } else { + Ok(CloseProgress::Pending) + } + } + + fn poll_close(&mut self, cx: &mut Context<'_>) -> Poll> { + if self.slot.drained() { + self.release_connection(); + Poll::Ready(Ok(())) + } else { + self.slot.register_close_waker(cx.waker()); + if self.slot.drained() { + self.release_connection(); + Poll::Ready(Ok(())) + } else { + Poll::Pending + } + } + } +} + +impl Drop for SqliteResource { + fn drop(&mut self) { + self.release_connection(); + } +} + +/// Shared state between one SQLite worker, its [`SqliteOpDriver`] operation, +/// and [`poll_pending_op`] on the VM thread. +/// +/// The worker writes the terminal signal and guest-visible value; the driver +/// reflects the signal into the operation registry and the VM wrapper reads +/// the value after the registry drive returns terminal. +struct SqliteOpShared { + cancelled: AtomicBool, + worker_done: AtomicBool, + signal: Mutex>>, + value: Mutex>>, + waker: Mutex>, + quiescence_waker: Mutex>, + worker: Mutex>>, +} + +impl SqliteOpShared { + fn new() -> Self { + Self { + cancelled: AtomicBool::new(false), + worker_done: AtomicBool::new(false), + signal: Mutex::new(None), + value: Mutex::new(None), + waker: Mutex::new(None), + quiescence_waker: Mutex::new(None), + worker: Mutex::new(None), + } + } + + fn is_quiescent(&self) -> bool { + self.worker_done.load(Ordering::Acquire) + } + + fn mark_worker_done(&self) { + self.worker_done.store(true, Ordering::Release); + if let Some(waker) = self + .quiescence_waker + .lock() + .expect("sqlite quiescence waker lock") + .take() + { + waker.wake(); + } + } + + fn register_quiescence_waker(&self, waker: &Waker) { + let mut guard = self + .quiescence_waker + .lock() + .expect("sqlite quiescence waker lock"); + if self.is_quiescent() { + return; + } + *guard = Some(waker.clone()); + if self.is_quiescent() + && let Some(waker) = guard.take() + { + waker.wake(); + } + } + + fn set_worker(&self, worker: JoinHandle<()>) { + *self.worker.lock().expect("sqlite worker lock") = Some(worker); + } + + fn join_worker(&self) -> bool { + self.worker + .lock() + .expect("sqlite worker lock") + .take() + .is_some_and(|worker| worker.join().is_err()) + } + + fn is_cancelled(&self) -> bool { + self.cancelled.load(Ordering::SeqCst) + } + + fn publish(&self, signal: Result<(), String>) { + *self.signal.lock().expect("sqlite signal lock") = Some(signal); + if let Some(waker) = self.waker.lock().expect("sqlite waker lock").take() { + waker.wake(); + } + } + + fn take_signal(&self) -> Option> { + self.signal.lock().expect("sqlite signal lock").take() + } + + fn register_waker(&self, waker: &Waker) { + *self.waker.lock().expect("sqlite waker lock") = Some(waker.clone()); + } + + fn fail(&self, error: VmError) { + let message = error.to_string(); + *self.value.lock().expect("sqlite value lock") = Some(Err(error)); + self.publish(Err(message)); + } + + fn succeed(&self, value: CallReturn) { + *self.value.lock().expect("sqlite value lock") = Some(Ok(value)); + self.publish(Ok(())); + } +} + +/// A concrete [`HostOperation`] driver for one pending SQLite operation. +/// +/// The operation id is filled in by [`schedule_operation`] after +/// [`ExecutionScope::start_operation`](crate::vm::execution_scope::ExecutionScope::start_operation) +/// assigns it, because the registry allocates packed ids internally. The +/// shared cell is written exactly once, before the driver can be polled or +/// cancelled (the operation is registered with the driver already boxed, but +/// the registry only drives it once the scheduler returns). +struct SqliteOpDriver { + shared: Arc, + slot: Arc, + id: Arc>>, + name: String, +} + +impl SqliteOpDriver { + fn new( + shared: Arc, + slot: Arc, + name: impl Into, + ) -> Self { + Self { + shared, + slot, + id: Arc::new(Mutex::new(None)), + name: name.into(), + } + } + + fn worker_failed(&self, message: String) -> Poll> { + Poll::Ready(Err(OperationError::new( + OperationErrorCode::OperationDriverFailed, + "sqlite::operation", + message, + ))) + } +} + +impl HostOperation for SqliteOpDriver { + fn poll(&mut self, cx: &mut Context<'_>) -> Poll> { + if !self.shared.is_quiescent() { + self.shared.register_waker(cx.waker()); + self.shared.register_quiescence_waker(cx.waker()); + if !self.shared.is_quiescent() { + return Poll::Pending; + } + } + if self.shared.is_cancelled() { + return self.worker_failed(format!("{} was cancelled", self.name)); + } + match self.shared.take_signal() { + Some(Ok(())) => Poll::Ready(Ok(())), + Some(Err(message)) => self.worker_failed(message), + None => self.worker_failed(format!( + "{} worker terminated without a completion signal", + self.name + )), + } + } + + fn cancel(&mut self, _reason: OperationCancelReason) -> OperationResult<()> { + self.shared.cancelled.store(true, Ordering::Release); + // If this operation is the one currently executing on the connection, + // interrupt the statement so the worker aborts promptly. Interrupting a + // connection with no active statement is a harmless no-op. + let is_active = self + .id + .lock() + .expect("sqlite driver id lock") + .is_some_and(|id| { + *self + .slot + .active_operation + .lock() + .expect("sqlite active lock") + == Some(id) + }); + if self.slot.closed.load(Ordering::Acquire) || is_active { + self.slot.interrupt.interrupt(); + } + Ok(()) + } + + fn is_quiescent(&self) -> bool { + self.shared.is_quiescent() + } + + fn register_quiescence_waker(&mut self, cx: &Context<'_>) { + self.shared.register_quiescence_waker(cx.waker()); + } + + fn cancel_and_wait(&mut self, reason: OperationCancelReason) -> OperationResult<()> { + self.cancel(reason)?; + if self.shared.join_worker() { + return Err(OperationError::new( + OperationErrorCode::OperationDriverFailed, + "sqlite::operation", + format!("{} worker panicked while cancelling", self.name), + )); + } + Ok(()) + } +} + +impl Drop for SqliteOpDriver { + fn drop(&mut self) { + if !self.shared.is_quiescent() { + let _ = self.cancel(OperationCancelReason::VmDrop); + } + let _ = self.shared.join_worker(); + } +} + +/// The per-VM SQLite adapter runtime state, mirroring the IO subsystem. +/// +/// Lives in the execution scope's typed arena (accessed lazily through +/// `ExecutionScope::scope_state_or_insert_with`), so it follows the scope +/// lifecycle: it is destroyed on reset/drop and recreated fresh on next use. +/// It owns the completion mailboxes for pending operations and an +/// adapter-owned counter of live connections used to enforce +/// `max_connections`. The embedding policy (`SqlitePolicy`) is *persistent* +/// module state stored in the generic `ModuleStateStore`, so it survives +/// `reset_for_reuse` while this runtime state does not. +pub(crate) struct SqliteState { + pending_results: HashMap>, + /// Adapter-owned live connection count, shared with each + /// [`SqliteResource`] so `begin_close` can decrement it. Avoids a generic + /// by-type close helper. + pub(crate) open_connections: Arc, +} + +impl Default for SqliteState { + fn default() -> Self { + Self { + pending_results: HashMap::new(), + open_connections: Arc::new(AtomicUsize::new(0)), + } + } +} + +/// Returns the adapter-declared SQLite scope state (the per-op completion +/// mailbox and the live-connection counter), creating the empty default on +/// first access while the scope is Active. The state is owned by the +/// execution-scope arena, so it is destroyed with the scope on reset and +/// recreated lazily on next use. +fn sqlite_state(vm: &mut Vm) -> VmResult<&mut SqliteState> { + vm.execution_scope() + .scope_state_or_insert_with(SqliteState::default) + .map_err(|error| VmError::HostError(format!("sqlite scope state unavailable: {error}"))) +} + +/// The default SQLite embedding policy used when no policy has been +/// configured through [`SqliteHostExt::configure_sqlite`]. This is the +/// value `SqlitePolicy::default()` produces, constructed explicitly so it can +/// live in a `static` and serve as the fallback behind the persistent +/// module-state lookup. +static DEFAULT_POLICY: SqlitePolicy = SqlitePolicy { + database_root: None, + allow_unsafe_sql: false, + limits: SqliteLimits { + max_connections: 16, + max_statements: 128, + max_rows: 1_000, + max_columns: 128, + max_result_bytes: 4 * 1024 * 1024, + max_statement_bytes: 1024 * 1024, + max_parameters: 128, + max_parameter_bytes: 1024 * 1024, + max_pending_operations: 32, + max_transaction_ms: 5_000, + busy_timeout_ms: 5_000, + }, +}; + +/// Returns the current persistent SQLite embedding policy, falling back to the +/// adapter default when none was configured through [`SqliteHostExt`]. +fn current_policy(vm: &Vm) -> &SqlitePolicy { + vm.host + .get_module_state::() + .unwrap_or(&DEFAULT_POLICY) +} + +fn sqlite_error(error: rusqlite::Error) -> VmError { + let code = error + .sqlite_error() + .map(|value| value.extended_code.to_string()) + .unwrap_or_else(|| "non_sqlite".to_string()); + let name = error + .sqlite_error_code() + .map(|value| format!("{value:?}")) + .unwrap_or_else(|| "RusqliteError".to_string()); + VmError::HostError(format!("SQLite error {name} ({code}): {error}")) +} + +fn cancellation_message(shared: &SqliteOpShared) -> String { + if shared.is_cancelled() { + "SQLite operation cancelled".to_string() + } else { + "SQLite connection was closed".to_string() + } +} + +/// Cancels one pending SQLite operation through the execution scope. +pub(super) fn cancel_pending_op(vm: &mut Vm, op_id: HostOpId) { + let Ok(id) = OperationId::from_raw(op_id) else { + return; + }; + // Drop the completion mailbox from the adapter-declared scope state; the + // operation's driver is cancelled through the registry (which forwards to + // the driver's `cancel`). + if let Ok(state) = sqlite_state(vm) { + state.pending_results.remove(&op_id); + } + let _ = vm + .execution_scope() + .cancel_operation(id, OperationCancelReason::Requested); +} + +/// Polls one pending SQLite operation through the execution scope's operation +/// registry, delivering the worker's guest-visible value. +pub(super) fn poll_pending_op( + vm: &mut Vm, + op_id: HostOpId, + cx: &mut Context<'_>, +) -> Poll> { + let id = match OperationId::from_raw(op_id) { + Ok(id) => id, + Err(error) => { + return Poll::Ready(Err(VmError::HostError(format!( + "invalid builtin sqlite op {op_id}: {error}" + )))); + } + }; + + let poll_result = vm.execution_scope().poll_operation(id, cx); + match poll_result { + Poll::Pending => Poll::Pending, + Poll::Ready(Err(error)) => { + // Drop the completion mailbox from the adapter-declared scope + // state; the operation itself already failed terminal. + if let Ok(state) = sqlite_state(vm) { + state.pending_results.remove(&op_id); + } + Poll::Ready(Err(VmError::HostError(format!( + "builtin sqlite op {op_id} failed: {error}" + )))) + } + Poll::Ready(Ok(outcome)) => { + // The worker wrote the authoritative guest-visible result into + // the completion mailbox before signalling terminal. Extract the + // mailbox entry (an `Arc`) so the scope borrow ends before any + // later scope mutation. + let shared = match sqlite_state(vm) { + Ok(state) => { + let Some(shared) = state.pending_results.remove(&op_id) else { + return Poll::Ready(Err(VmError::HostError(format!( + "builtin sqlite op {op_id} has no completion mailbox" + )))); + }; + shared + } + Err(error) => { + return Poll::Ready(Err(VmError::HostError(format!( + "builtin sqlite op {op_id} mailbox unavailable: {error}" + )))); + } + }; + // A cancelled/closed operation reports a guest-visible error even if + // the worker happened to complete concurrently. + if matches!( + outcome, + crate::vm::operation::driver::OperationOutcome::Cancelled(_) + ) || shared.is_cancelled() + { + return Poll::Ready(Err(VmError::HostError(cancellation_message(&shared)))); + } + let value = shared.value.lock().expect("sqlite value lock").take(); + match value { + Some(value) => Poll::Ready(value), + None => Poll::Ready(Err(VmError::HostError(format!( + "builtin sqlite op {op_id} completed without a result" + )))), + } + } + } +} + +fn handle_value(handle: ResourceHandle) -> i64 { + handle.raw() as i64 +} + +fn sqlite_handle(handle_id: i64) -> VmResult { + if handle_id <= 0 { + return Err(VmError::HostError(format!( + "invalid sqlite handle id {handle_id}; expected positive handle id" + ))); + } + ResourceHandle::from_raw(handle_id as u64).map_err(|error| { + VmError::HostError(format!("invalid sqlite handle id {handle_id}: {error}")) + }) +} + +/// Lifts a guest-visible integer handle into a typed, live scope token. +/// +/// This validates arena, slot, generation, open state, and `TypeId` through +/// the generic typed table — a foreign, stale, closed, or wrong-typed handle +/// is rejected here before any SQLite state is touched. +fn lookup_connection(vm: &mut Vm, handle_id: i64) -> VmResult> { + let handle = sqlite_handle(handle_id)?; + let token = vm + .execution_scope() + .resources() + .typed::(handle) + .map_err(|error| VmError::HostError(format!("unknown SQLite database: {error}")))?; + let resource = vm + .execution_scope() + .resources() + .get::(&token) + .map_err(|error| VmError::HostError(format!("SQLite database borrow failed: {error}")))?; + if resource.slot.closed.load(Ordering::SeqCst) { + return Err(VmError::HostError( + "SQLite database is already closed".to_string(), + )); + } + Ok(Arc::clone(&resource.slot)) +} + +fn map_value<'a>(map: &'a VmMap, key: &str) -> Option<&'a Value> { + map.get(&Value::string(key)) +} + +fn required_string(map: &VmMap, key: &str) -> VmResult { + match map_value(map, key) { + Some(Value::String(value)) if !value.is_empty() => Ok(value.as_ref().clone()), + Some(Value::String(_)) => Err(VmError::HostError(format!( + "SQLite {key} must not be empty" + ))), + Some(_) => Err(VmError::TypeMismatch("SQLite option string")), + None => Err(VmError::HostError(format!("missing SQLite {key}"))), + } +} + +fn optional_string(map: &VmMap, key: &str) -> VmResult> { + match map_value(map, key) { + Some(Value::String(value)) => Ok(Some(value.as_ref().clone())), + Some(Value::Null) | None => Ok(None), + Some(_) => Err(VmError::TypeMismatch("SQLite option string")), + } +} + +fn parse_positive_usize(value: &Value, label: &str) -> VmResult { + let Value::Int(value) = value else { + return Err(VmError::TypeMismatch("SQLite limit integer")); + }; + if *value <= 0 { + return Err(VmError::HostError(format!( + "SQLite {label} must be positive" + ))); + } + usize::try_from(*value).map_err(|_| VmError::HostError(format!("SQLite {label} is too large"))) +} + +fn parse_positive_u64(value: &Value, label: &str) -> VmResult { + let Value::Int(value) = value else { + return Err(VmError::TypeMismatch("SQLite limit integer")); + }; + if *value <= 0 { + return Err(VmError::HostError(format!( + "SQLite {label} must be positive" + ))); + } + u64::try_from(*value).map_err(|_| VmError::HostError(format!("SQLite {label} is too large"))) +} + +fn parse_limits(value: Option<&Value>, ceiling: SqliteLimits) -> VmResult { + let Some(value) = value else { + return Ok(ceiling); + }; + let Value::Map(map) = value else { + return Err(VmError::TypeMismatch("SQLite limits map")); + }; + let mut limits = ceiling; + for (key, value) in map.iter() { + let Value::String(key) = key else { + return Err(VmError::TypeMismatch("SQLite limit name")); + }; + match key.as_str() { + "max_connections" => { + limits.max_connections = + parse_positive_usize(value, key)?.min(ceiling.max_connections) + } + "max_statements" => { + limits.max_statements = + parse_positive_usize(value, key)?.min(ceiling.max_statements) + } + "max_rows" => limits.max_rows = parse_positive_usize(value, key)?.min(ceiling.max_rows), + "max_columns" => { + limits.max_columns = parse_positive_usize(value, key)?.min(ceiling.max_columns) + } + "max_result_bytes" => { + limits.max_result_bytes = + parse_positive_usize(value, key)?.min(ceiling.max_result_bytes) + } + "max_statement_bytes" => { + limits.max_statement_bytes = + parse_positive_usize(value, key)?.min(ceiling.max_statement_bytes) + } + "max_parameters" => { + limits.max_parameters = + parse_positive_usize(value, key)?.min(ceiling.max_parameters) + } + "max_parameter_bytes" => { + limits.max_parameter_bytes = + parse_positive_usize(value, key)?.min(ceiling.max_parameter_bytes) + } + "max_pending_operations" => { + limits.max_pending_operations = + parse_positive_usize(value, key)?.min(ceiling.max_pending_operations) + } + "max_transaction_ms" => { + limits.max_transaction_ms = + parse_positive_u64(value, key)?.min(ceiling.max_transaction_ms) + } + "busy_timeout_ms" => { + limits.busy_timeout_ms = + parse_positive_u64(value, key)?.min(ceiling.busy_timeout_ms) + } + _ => { + return Err(VmError::HostError(format!("unknown SQLite limit {key}"))); + } + } + } + Ok(limits) +} + +fn parse_query_limits(value: &VmMap, ceiling: SqliteLimits) -> VmResult { + parse_limits(Some(&Value::Map(Arc::new(value.clone()))), ceiling) +} + +fn validate_relative_path(path: &Path) -> VmResult<()> { + if path.as_os_str().is_empty() || path.is_absolute() { + return Err(VmError::HostError( + "SQLite database path must be a non-empty relative path".to_string(), + )); + } + if path.components().any(|component| { + matches!( + component, + Component::ParentDir | Component::RootDir | Component::Prefix(_) + ) + }) { + return Err(VmError::HostError( + "SQLite database path must stay below its configured root".to_string(), + )); + } + Ok(()) +} + +fn canonical_root(root: &Path) -> VmResult { + if !root.is_absolute() { + return Err(VmError::HostError( + "SQLite database root must be absolute".to_string(), + )); + } + fs::canonicalize(root) + .map_err(|error| VmError::HostError(format!("invalid SQLite database root: {error}"))) +} + +fn resolve_database_path(options: &OpenOptions) -> VmResult> { + if options.mode == OpenMode::Memory { + if options.path != ":memory:" { + return Err(VmError::HostError( + "SQLite memory mode requires path ':memory:'".to_string(), + )); + } + return Ok(None); + } + if options.path == ":memory:" { + return Err(VmError::HostError( + "SQLite ':memory:' requires memory open mode".to_string(), + )); + } + let root = options + .root + .as_deref() + .ok_or_else(|| VmError::HostError("SQLite database root is required".to_string()))?; + let root = canonical_root(root)?; + let relative = Path::new(&options.path); + validate_relative_path(relative)?; + let candidate = root.join(relative); + let canonical = if candidate.exists() { + fs::canonicalize(&candidate) + .map_err(|error| VmError::HostError(format!("invalid SQLite database path: {error}")))? + } else { + if options.mode != OpenMode::ReadWriteCreate { + return Err(VmError::HostError(format!( + "SQLite database does not exist: {}", + candidate.display() + ))); + } + let parent = candidate + .parent() + .ok_or_else(|| VmError::HostError("SQLite database path has no parent".to_string()))?; + let canonical_parent = fs::canonicalize(parent).map_err(|error| { + VmError::HostError(format!("invalid SQLite database parent: {error}")) + })?; + let file_name = candidate.file_name().ok_or_else(|| { + VmError::HostError("SQLite database path has no file name".to_string()) + })?; + canonical_parent.join(file_name) + }; + if !canonical.starts_with(&root) { + return Err(VmError::HostError( + "SQLite database path escapes its configured root".to_string(), + )); + } + Ok(Some(canonical)) +} + +fn sqlite_limit(value: usize, label: &str) -> VmResult { + i32::try_from(value) + .map_err(|_| VmError::HostError(format!("SQLite {label} exceeds engine limits"))) +} + +fn install_connection_limits(connection: &Connection, limits: SqliteLimits) -> VmResult<()> { + let max_value_bytes = limits.max_result_bytes.max(limits.max_parameter_bytes); + connection.set_limit( + Limit::SQLITE_LIMIT_LENGTH, + sqlite_limit(max_value_bytes, "value byte limit")?, + ); + connection.set_limit( + Limit::SQLITE_LIMIT_SQL_LENGTH, + sqlite_limit(limits.max_statement_bytes, "statement byte limit")?, + ); + connection.set_limit( + Limit::SQLITE_LIMIT_COLUMN, + sqlite_limit(limits.max_columns, "column limit")?, + ); + connection.set_limit( + Limit::SQLITE_LIMIT_VARIABLE_NUMBER, + sqlite_limit(limits.max_parameters, "parameter count limit")?, + ); + Ok(()) +} + +fn install_authorizer(connection: &Connection, allow_unsafe_sql: bool) { + connection.authorizer(Some(move |context: AuthContext<'_>| { + if allow_unsafe_sql { + return Authorization::Allow; + } + match context.action { + AuthAction::Attach { .. } + | AuthAction::Detach { .. } + | AuthAction::Pragma { .. } + | AuthAction::CreateVtable { .. } + | AuthAction::DropVtable { .. } + | AuthAction::Unknown { .. } => Authorization::Deny, + AuthAction::Function { function_name } + if function_name.eq_ignore_ascii_case("load_extension") => + { + Authorization::Deny + } + _ => Authorization::Allow, + } + })); +} + +fn open_connection(options: &OpenOptions) -> VmResult { + let path = resolve_database_path(options)?; + let flags = match options.mode { + OpenMode::Memory => OpenFlags::SQLITE_OPEN_READ_WRITE | OpenFlags::SQLITE_OPEN_CREATE, + OpenMode::ReadOnly => OpenFlags::SQLITE_OPEN_READ_ONLY, + OpenMode::ReadWrite => OpenFlags::SQLITE_OPEN_READ_WRITE, + OpenMode::ReadWriteCreate => { + OpenFlags::SQLITE_OPEN_READ_WRITE | OpenFlags::SQLITE_OPEN_CREATE + } + } | OpenFlags::SQLITE_OPEN_NO_MUTEX; + let connection = match path { + Some(path) => Connection::open_with_flags(path, flags), + None => Connection::open_in_memory_with_flags(flags), + } + .map_err(sqlite_error)?; + connection + .busy_timeout(Duration::from_millis(options.limits.busy_timeout_ms)) + .map_err(sqlite_error)?; + install_connection_limits(&connection, options.limits)?; + install_authorizer(&connection, options.allow_unsafe_sql); + Ok(connection) +} + +fn normalized_sql(sql: &str) -> VmResult { + let bytes = sql.as_bytes(); + let mut out = String::with_capacity(sql.len()); + let mut index = 0; + let mut quote = None; + let mut statement_ended = false; + while index < bytes.len() { + let byte = bytes[index]; + if let Some(active_quote) = quote { + if byte == active_quote { + if index + 1 < bytes.len() && bytes[index + 1] == active_quote { + index += 2; + continue; + } + quote = None; + } + index += 1; + continue; + } + if matches!(byte, b'\'' | b'"' | b'`') { + quote = Some(byte); + out.push(' '); + index += 1; + continue; + } + if byte == b'-' && index + 1 < bytes.len() && bytes[index + 1] == b'-' { + index += 2; + while index < bytes.len() && bytes[index] != b'\n' { + index += 1; + } + out.push(' '); + continue; + } + if byte == b'/' && index + 1 < bytes.len() && bytes[index + 1] == b'*' { + index += 2; + while index + 1 < bytes.len() && !(bytes[index] == b'*' && bytes[index + 1] == b'/') { + index += 1; + } + if index + 1 >= bytes.len() { + return Err(VmError::HostError( + "SQLite SQL contains an unterminated comment".to_string(), + )); + } + index += 2; + out.push(' '); + continue; + } + if byte == b';' { + statement_ended = true; + index += 1; + continue; + } + if statement_ended && !byte.is_ascii_whitespace() { + return Err(VmError::HostError( + "multiple SQLite statements are not allowed".to_string(), + )); + } + out.push((byte as char).to_ascii_lowercase()); + index += 1; + } + if quote.is_some() { + return Err(VmError::HostError( + "SQLite SQL contains an unterminated quote".to_string(), + )); + } + Ok(out) +} + +fn validate_sql(sql: &str, limits: SqliteLimits, allow_unsafe_sql: bool) -> VmResult<()> { + if sql.is_empty() || sql.len() > limits.max_statement_bytes || sql.as_bytes().contains(&0) { + return Err(VmError::HostError(format!( + "SQLite statement exceeds the configured {} byte limit or is invalid", + limits.max_statement_bytes + ))); + } + let normalized = normalized_sql(sql)?; + if allow_unsafe_sql { + return Ok(()); + } + let first = normalized.split_whitespace().next().unwrap_or_default(); + if matches!( + first, + "attach" + | "detach" + | "pragma" + | "vacuum" + | "begin" + | "commit" + | "rollback" + | "savepoint" + | "release" + ) { + return Err(VmError::HostError(format!( + "SQLite statement {first} is not allowed" + ))); + } + if normalized + .split(|character: char| !character.is_ascii_alphanumeric() && character != '_') + .any(|token| token == "load_extension") + { + return Err(VmError::HostError( + "SQLite extension loading is disabled".to_string(), + )); + } + Ok(()) +} + +fn sqlite_params(values: VmArrayRef<'_>, limits: SqliteLimits) -> VmResult> { + if values.len() > limits.max_parameters { + return Err(VmError::HostError( + "SQLite parameter count exceeds the configured limit".to_string(), + )); + } + let mut bytes = 0usize; + let mut params = Vec::with_capacity(values.len()); + for value in values { + let sql_value = match value { + Value::Null => SqlValue::Null, + Value::Int(value) => SqlValue::Integer(*value), + Value::Float(value) => SqlValue::Real(*value), + Value::String(value) => { + bytes = bytes.saturating_add(value.len()); + SqlValue::Text(value.as_ref().clone()) + } + Value::Bytes(value) => { + bytes = bytes.saturating_add(value.len()); + SqlValue::Blob(value.as_ref().clone()) + } + _ => { + return Err(VmError::HostError( + "SQLite parameters support only null, int, float, string, and bytes" + .to_string(), + )); + } + }; + if bytes > limits.max_parameter_bytes { + return Err(VmError::HostError(format!( + "SQLite parameters exceed the configured {} byte limit", + limits.max_parameter_bytes + ))); + } + params.push(sql_value); + } + Ok(params) +} + +/// Runs one synchronous closure against the connection, with cancellation +/// surfaced through the shared cancelled flag. +fn with_connection( + slot: &ConnectionSlot, + shared: &Arc, + operation: impl FnOnce(&mut Connection) -> Result, +) -> VmResult { + if slot.closed.load(Ordering::Acquire) || shared.is_cancelled() { + return Err(VmError::HostError(cancellation_message(shared))); + } + let mut connection = slot + .connection + .lock() + .map_err(|_| VmError::HostError("SQLite connection lock is poisoned".to_string()))?; + if slot.closed.load(Ordering::Acquire) || shared.is_cancelled() { + return Err(VmError::HostError(cancellation_message(shared))); + } + let handler_shared = Arc::clone(shared); + connection.progress_handler( + SQLITE_PROGRESS_STEPS, + Some(move || handler_shared.is_cancelled()), + ); + let result = operation(&mut connection); + connection.progress_handler(0, None:: bool>); + if slot.closed.load(Ordering::Acquire) || shared.is_cancelled() { + return Err(VmError::HostError(cancellation_message(shared))); + } + result.map_err(sqlite_error) +} + +fn estimate_value_bytes(value: &Value) -> usize { + match value { + Value::Null => 1, + Value::Int(_) | Value::Float(_) => 8, + Value::Bool(_) => 1, + Value::String(value) => value.len(), + Value::Bytes(value) => value.len(), + Value::Array(values) => values.iter().map(estimate_value_bytes).sum(), + Value::Map(values) => values + .iter() + .map(|(key, value)| { + estimate_value_bytes(key).saturating_add(estimate_value_bytes(value)) + }) + .sum(), + Value::Callable(_) => 8, + } +} + +fn value_from_row(row: &rusqlite::Row<'_>, index: usize) -> Result { + match row.get_ref(index)? { + ValueRef::Null => Ok(Value::Null), + ValueRef::Integer(value) => Ok(Value::Int(value)), + ValueRef::Real(value) => Ok(Value::Float(value)), + ValueRef::Text(value) => match std::str::from_utf8(value) { + Ok(value) => Ok(Value::string(value)), + Err(_) => Ok(Value::bytes(value.to_vec())), + }, + ValueRef::Blob(value) => Ok(Value::bytes(value.to_vec())), + } +} + +fn query_with_connection( + connection: &Connection, + sql: &str, + params: &[SqlValue], + limits: SqliteLimits, +) -> Result { + let mut statement = connection.prepare(sql)?; + let columns = statement + .column_names() + .into_iter() + .map(Value::string) + .collect::>(); + if columns.len() > limits.max_columns { + return Err(rusqlite::Error::InvalidColumnIndex(columns.len())); + } + let column_count = columns.len(); + let mut rows = statement.query(params_from_iter(params.iter()))?; + let mut values = Vec::new(); + let mut result_bytes = columns.iter().map(estimate_value_bytes).sum::(); + let mut truncated = false; + let mut next_cursor = None; + while let Some(row) = rows.next()? { + if values.len() >= limits.max_rows { + truncated = true; + break; + } + let mut cells = Vec::with_capacity(column_count); + let mut row_bytes = 0usize; + for index in 0..column_count { + let value = value_from_row(row, index)?; + row_bytes = row_bytes.saturating_add(estimate_value_bytes(&value)); + cells.push(value); + } + if result_bytes.saturating_add(row_bytes) > limits.max_result_bytes { + truncated = true; + break; + } + if let Some(Value::Int(cursor)) = cells.first() { + next_cursor = Some(*cursor); + } + result_bytes = result_bytes.saturating_add(row_bytes); + values.push(Value::array(cells)); + } + let mut entries = vec![ + (Value::string("columns"), Value::array(columns)), + (Value::string("rows"), Value::array(values)), + (Value::string("truncated"), Value::Bool(truncated)), + ]; + if let Some(next_cursor) = next_cursor { + entries.push((Value::string("next_cursor"), Value::Int(next_cursor))); + } + Ok(VmMap::from_entries(entries)) +} + +fn execute_with_connection( + connection: &Connection, + sql: &str, + params: &[SqlValue], +) -> Result { + let mut statement = connection.prepare(sql)?; + let rows_affected = statement.execute(params_from_iter(params.iter()))?; + drop(statement); + Ok(VmMap::from_entries(vec![ + ( + Value::string("rows_affected"), + Value::Int(i64::try_from(rows_affected).unwrap_or(i64::MAX)), + ), + ( + Value::string("last_insert_rowid"), + Value::Int(connection.last_insert_rowid()), + ), + ])) +} + +struct SqliteWorkerCompletion { + slot: Arc, + shared: Arc, + id: OperationId, +} + +impl Drop for SqliteWorkerCompletion { + fn drop(&mut self) { + if let Ok(mut active) = self.slot.active_operation.lock() + && *active == Some(self.id) + { + *active = None; + } + self.shared.mark_worker_done(); + self.slot.unregister(self.id); + } +} + +/// Schedules a worker thread to run one SQLite operation on a connection and +/// registers its [`SqliteOpDriver`] in the VM's execution scope. +/// +/// The driver is constructed with a shared id cell that +/// [`ExecutionScope::start_operation`](crate::vm::execution_scope::ExecutionScope::start_operation) +/// fills in after allocating the packed operation id, so the driver's `cancel` +/// can compare against the connection's active operation without a registry +/// fixup. The worker holds the connection's execution mutex for the whole +/// operation (serializing access, since SQLite connections are not +/// thread-safe), records itself as the active operation, and publishes the +/// terminal signal plus the guest-visible value through the shared mailbox. +fn schedule_operation( + vm: &mut Vm, + slot: Arc, + operation: impl FnOnce(Arc, Arc) -> VmResult + + Send + + 'static, +) -> VmResult { + if slot.closed.load(Ordering::SeqCst) { + return Err(VmError::HostError( + "SQLite database is already closed".to_string(), + )); + } + if slot.pending_count() >= slot.limits.max_pending_operations { + return Err(VmError::HostError(format!( + "SQLite pending operation limit {} reached", + slot.limits.max_pending_operations + ))); + } + + let shared = Arc::new(SqliteOpShared::new()); + let worker_shared = Arc::clone(&shared); + let worker_slot = Arc::clone(&slot); + let worker_name = "sqlite::operation".to_string(); + let driver = SqliteOpDriver::new(Arc::clone(&shared), Arc::clone(&slot), worker_name.clone()); + let driver_id = Arc::clone(&driver.id); + + let deadline = + Instant::now().checked_add(Duration::from_millis(slot.limits.max_transaction_ms)); + let spec = OperationSpec::new(driver) + .with_deadline(deadline.unwrap_or_else(|| Instant::now() + Duration::from_secs(3600))); + + let op_id = vm + .execution_scope() + .start_operation(spec) + .map_err(|error| { + VmError::HostError(format!("failed to start sqlite operation: {error}")) + })?; + *driver_id + .lock() + .expect("sqlite driver id lock should not be poisoned") = Some(op_id); + slot.register(op_id); + let raw = op_id.raw(); + + let worker = thread::Builder::new() + .name(format!("rustscript-sqlite-{raw}")) + .spawn(move || { + let _completion = SqliteWorkerCompletion { + slot: Arc::clone(&worker_slot), + shared: Arc::clone(&worker_shared), + id: op_id, + }; + let _execution = worker_slot + .execution + .lock() + .expect("SQLite execution lock should not be poisoned"); + if worker_slot.closed.load(Ordering::Acquire) || worker_shared.is_cancelled() { + worker_shared.fail(VmError::HostError(cancellation_message(&worker_shared))); + return; + } + *worker_slot + .active_operation + .lock() + .expect("SQLite active operation lock should not be poisoned") = Some(op_id); + if worker_slot.closed.load(Ordering::Acquire) || worker_shared.is_cancelled() { + worker_shared.fail(VmError::HostError(cancellation_message(&worker_shared))); + return; + } + let result = operation(Arc::clone(&worker_slot), Arc::clone(&worker_shared)); + match result { + Ok(value) => worker_shared.succeed(value), + Err(error) => worker_shared.fail(error), + } + }) + .map_err(|error| { + shared.mark_worker_done(); + let _ = vm + .execution_scope() + .abort_operation(op_id, OperationCancelReason::Requested); + slot.unregister(op_id); + VmError::HostError(format!("failed to spawn sqlite worker: {error}")) + })?; + shared.set_worker(worker); + + sqlite_state(vm)?.pending_results.insert(raw, shared); + Ok(raw) +} + +/// Parses the `sqlite::open` options map against the adapter-owned embedding +/// policy. +fn parse_open_options(vm: &Vm, options: &VmMap) -> VmResult { + let policy = current_policy(vm); + let path = required_string(options, "path")?; + let mode = match optional_string(options, "mode")?.as_deref() { + Some("memory") => OpenMode::Memory, + Some("read_only") => OpenMode::ReadOnly, + Some("read_write") => OpenMode::ReadWrite, + Some("read_write_create") | None => OpenMode::ReadWriteCreate, + Some(mode) => { + return Err(VmError::HostError(format!( + "unknown SQLite open mode {mode}" + ))); + } + }; + let configured_root = policy.database_root.as_deref().map(PathBuf::from); + if let Some(requested_root) = optional_string(options, "root")? { + let requested_root = PathBuf::from(requested_root); + if configured_root.as_ref() != Some(&requested_root) { + return Err(VmError::HostError( + "SQLite root must match the embedding policy".to_string(), + )); + } + } + if mode != OpenMode::Memory && configured_root.is_none() { + return Err(VmError::HostError( + "SQLite database root is not configured".to_string(), + )); + } + let limits = parse_limits(map_value(options, "limits"), policy.limits)?; + Ok(OpenOptions { + path, + mode, + root: configured_root, + limits, + allow_unsafe_sql: policy.allow_unsafe_sql, + }) +} + +/// Opens a SQLite database under the embedding-owned path and limit policy. +/// +/// The connection is stored as a typed [`SqliteResource`] in the execution +/// scope; the guest-visible handle is the raw scope handle, validated for +/// arena, slot, generation, open state, and type on every later use. The +/// live-connection count is adapter-owned (shared with each resource) so +/// `max_connections` is enforced without a generic by-type helper. +#[pd_host_function(name = "sqlite::open")] +pub(super) fn builtin_sqlite_open_impl(vm: &mut Vm, options: VmMapRef<'_>) -> VmResult { + let options = parse_open_options(vm, options)?; + // The adapter-declared scope state owns the live-connection counter; + // clone the `Arc` so the scope borrow ends before `push_resource` below. + let open_connections: Arc = Arc::clone(&sqlite_state(vm)?.open_connections); + if open_connections.load(Ordering::SeqCst) >= options.limits.max_connections { + return Err(VmError::HostError(format!( + "SQLite connection limit {} reached", + options.limits.max_connections + ))); + } + let connection = open_connection(&options)?; + let interrupt = connection.get_interrupt_handle(); + let slot = Arc::new(ConnectionSlot { + connection: Mutex::new(connection), + execution: Mutex::new(()), + active_operation: Mutex::new(None), + pending: Mutex::new(Vec::new()), + live_workers: AtomicUsize::new(0), + close_waker: Mutex::new(None), + interrupt: Arc::new(interrupt), + limits: options.limits, + allow_unsafe_sql: options.allow_unsafe_sql, + closed: AtomicBool::new(false), + }); + let resource = vm + .execution_scope() + .push_resource(SqliteResource::new(slot, Arc::clone(&open_connections))) + .map_err(|error| VmError::HostError(format!("failed to open SQLite database: {error}")))?; + open_connections.fetch_add(1, Ordering::SeqCst); + Ok(handle_value(resource.handle())) +} + +/// Executes one parameterized SQLite statement asynchronously. +#[pd_host_function(name = "sqlite::execute")] +pub(super) fn builtin_sqlite_execute_impl( + vm: &mut Vm, + db_id: i64, + sql: &str, + params: VmArrayRef<'_>, +) -> VmResult> { + let slot = lookup_connection(vm, db_id)?; + validate_sql(sql, slot.limits, slot.allow_unsafe_sql)?; + let sql = sql.to_string(); + let params = sqlite_params(params, slot.limits)?; + let op_id = schedule_operation(vm, slot, move |slot, shared| { + with_connection(&slot, &shared, |connection| { + execute_with_connection(connection, &sql, ¶ms) + }) + .map(|value| CallReturn::one(Value::Map(Arc::new(value)))) + })?; + Ok(HostCallResult::Pending(op_id)) +} + +/// Runs one parameterized SQLite query with row and result-byte bounds. +#[pd_host_function(name = "sqlite::query")] +pub(super) fn builtin_sqlite_query_impl( + vm: &mut Vm, + db_id: i64, + sql: &str, + params: VmArrayRef<'_>, + limits: VmMapRef<'_>, +) -> VmResult> { + let slot = lookup_connection(vm, db_id)?; + let query_limits = parse_query_limits(limits, slot.limits)?; + validate_sql(sql, query_limits, slot.allow_unsafe_sql)?; + let sql = sql.to_string(); + let params = sqlite_params(params, slot.limits)?; + let op_id = schedule_operation(vm, slot, move |slot, shared| { + with_connection(&slot, &shared, |connection| { + query_with_connection(connection, &sql, ¶ms, query_limits) + }) + .map(|value| CallReturn::one(Value::Map(Arc::new(value)))) + })?; + Ok(HostCallResult::Pending(op_id)) +} + +struct TransactionStatement { + sql: String, + params: Vec, + query: bool, + limits: SqliteLimits, +} + +fn parse_transaction_statements( + statements: VmArrayRef<'_>, + limits: SqliteLimits, + allow_unsafe_sql: bool, +) -> VmResult> { + if statements.is_empty() { + return Err(VmError::HostError( + "SQLite transaction requires at least one statement".to_string(), + )); + } + if statements.len() > limits.max_statements { + return Err(VmError::HostError(format!( + "SQLite transaction exceeds the configured {} statement limit", + limits.max_statements + ))); + } + statements + .iter() + .map(|statement| { + let Value::Map(statement) = statement else { + return Err(VmError::TypeMismatch("SQLite transaction statement map")); + }; + let sql = required_string(statement, "sql")?; + validate_sql(&sql, limits, allow_unsafe_sql)?; + let params = match map_value(statement, "params") { + Some(Value::Array(params)) => sqlite_params(params, limits)?, + Some(_) => return Err(VmError::TypeMismatch("SQLite parameter array")), + None => Vec::new(), + }; + let query = match map_value(statement, "query") { + Some(Value::Bool(query)) => *query, + Some(_) => return Err(VmError::TypeMismatch("SQLite query flag")), + None => false, + }; + let statement_limits = match map_value(statement, "limits") { + Some(Value::Map(statement_limits)) => parse_query_limits(statement_limits, limits)?, + Some(_) => return Err(VmError::TypeMismatch("SQLite limits map")), + None => limits, + }; + Ok(TransactionStatement { + sql, + params, + query, + limits: statement_limits, + }) + }) + .collect() +} + +/// Runs ordered statements atomically and returns ordered result envelopes. +#[pd_host_function(name = "sqlite::transaction")] +pub(super) fn builtin_sqlite_transaction_impl( + vm: &mut Vm, + db_id: i64, + statements: VmArrayRef<'_>, +) -> VmResult>> { + let slot = lookup_connection(vm, db_id)?; + let statements = parse_transaction_statements(statements, slot.limits, slot.allow_unsafe_sql)?; + let op_id = schedule_operation(vm, slot, move |slot, shared| { + with_connection(&slot, &shared, |connection| { + let transaction = + connection.transaction_with_behavior(TransactionBehavior::Immediate)?; + let mut results = Vec::with_capacity(statements.len()); + for statement in statements { + let value = if statement.query { + query_with_connection( + &transaction, + &statement.sql, + &statement.params, + statement.limits, + )? + } else { + execute_with_connection(&transaction, &statement.sql, &statement.params)? + }; + results.push(Value::Map(Arc::new(value))); + } + transaction.commit()?; + Ok(results) + }) + .map(|values| CallReturn::one(Value::array(values))) + })?; + Ok(HostCallResult::Pending(op_id)) +} + +/// Closes a SQLite resource through the generic scope close. Pending drivers +/// on the connection observe the closed slot and are retired through the +/// scope's operation registry; no type-dispatched helper is needed. +#[pd_host_function(name = "sqlite::close")] +pub(super) fn builtin_sqlite_close_impl(vm: &mut Vm, db_id: i64) -> VmResult<()> { + let handle = sqlite_handle(db_id)?; + vm.execution_scope() + .close_resource::(handle, ResourceCloseReason::Requested) + .map_err(|error| VmError::HostError(format!("unknown SQLite database: {error}")))?; + Ok(()) +} + +/// Adapter-owned SQLite embedding-control surface. +/// +/// The concrete SQLite `configure`/`clear`/policy-read control API lives in +/// this adapter module (implemented on the public [`Vm`]) rather than on the +/// generic `vm` layer, so `src/vm/**` never names the SQLite policy type or a +/// concrete control method. Replaces `configure_policy` / `clear_policy` +/// free functions and the SQLite methods that used to live in `src/vm/mod.rs`. +pub trait SqliteHostExt { + /// Replaces the adapter-owned SQLite embedding policy. + /// + /// Open connections keep the limits they were opened with; new opens use + /// this policy. The policy is stored in the persistent, reset-surviving + /// `ModuleStateStore`: it stays in force across `reset_for_reuse` while + /// the adapter's per-invocation scope state is destroyed and recreated. + fn configure_sqlite(&mut self, policy: SqlitePolicy); + + /// Restores the default SQLite embedding policy. + /// + /// Pending operations and open connections are unaffected (they carry + /// their own state); a VM reset or explicit `sqlite::close` retires them + /// through the generic scope close. The persistent policy entry is + /// removed, so subsequent opens fall back to the adapter default. + fn clear_sqlite(&mut self); + + /// Returns the current SQLite embedding policy. + fn sqlite_policy(&self) -> &SqlitePolicy; +} + +impl SqliteHostExt for Vm { + fn configure_sqlite(&mut self, policy: SqlitePolicy) { + // Adapter-declared policy stored in the generic module-state store: + // module-level policy survives execution-scope reset (an embedder's + // root/limits remain in force across `reset_for_reuse`), while the + // adapter's per-invocation runtime state lives in the scope arena. + self.host.set_module_state(policy); + } + + fn clear_sqlite(&mut self) { + self.host.remove_module_state::(); + } + + fn sqlite_policy(&self) -> &SqlitePolicy { + current_policy(self) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn test_operation_id(slot: u64) -> OperationId { + OperationId::from_raw((1 << 43) | (slot << 22) | 1).expect("valid test operation id") + } + + #[test] + fn close_waits_for_active_and_queued_workers() { + let connection = Connection::open_in_memory().expect("in-memory SQLite connection"); + let interrupt = connection.get_interrupt_handle(); + let slot = Arc::new(ConnectionSlot { + connection: Mutex::new(connection), + execution: Mutex::new(()), + active_operation: Mutex::new(None), + pending: Mutex::new(Vec::new()), + live_workers: AtomicUsize::new(0), + close_waker: Mutex::new(None), + interrupt: Arc::new(interrupt), + limits: SqliteLimits::default(), + allow_unsafe_sql: false, + closed: AtomicBool::new(false), + }); + let open_connections = Arc::new(AtomicUsize::new(1)); + let mut resource = SqliteResource::new(Arc::clone(&slot), Arc::clone(&open_connections)); + let active_id = test_operation_id(1); + let queued_id = test_operation_id(2); + slot.register(active_id); + slot.register(queued_id); + + let release_active = Arc::new(AtomicBool::new(false)); + let active_started = Arc::new(AtomicBool::new(false)); + let active_slot = Arc::clone(&slot); + let active_release = Arc::clone(&release_active); + let active_started_flag = Arc::clone(&active_started); + let active = thread::spawn(move || { + let _execution = active_slot.execution.lock().expect("execution lock"); + active_started_flag.store(true, Ordering::Release); + while !active_release.load(Ordering::Acquire) { + thread::yield_now(); + } + active_slot.unregister(active_id); + }); + while !active_started.load(Ordering::Acquire) { + thread::yield_now(); + } + + let queued_started = Arc::new(AtomicBool::new(false)); + let queued_executed = Arc::new(AtomicBool::new(false)); + let queued_slot = Arc::clone(&slot); + let queued_started_flag = Arc::clone(&queued_started); + let queued_executed_flag = Arc::clone(&queued_executed); + let queued = thread::spawn(move || { + queued_started_flag.store(true, Ordering::Release); + let _execution = queued_slot.execution.lock().expect("execution lock"); + if !queued_slot.closed.load(Ordering::Acquire) { + queued_executed_flag.store(true, Ordering::Release); + } + queued_slot.unregister(queued_id); + }); + while !queued_started.load(Ordering::Acquire) { + thread::yield_now(); + } + + assert_eq!( + resource + .begin_close(ResourceCloseReason::Requested) + .expect("close should begin"), + CloseProgress::Pending + ); + let mut cx = Context::from_waker(Waker::noop()); + assert!(matches!(resource.poll_close(&mut cx), Poll::Pending)); + assert!(!queued_executed.load(Ordering::Acquire)); + + release_active.store(true, Ordering::Release); + active.join().expect("active worker should finish"); + queued.join().expect("queued worker should finish"); + assert!(!queued_executed.load(Ordering::Acquire)); + assert!(slot.drained()); + assert!(matches!(resource.poll_close(&mut cx), Poll::Ready(Ok(())))); + assert_eq!(open_connections.load(Ordering::Acquire), 0); + } +} diff --git a/src/cli.rs b/src/cli.rs index 45684d7f..68d4bcf9 100644 --- a/src/cli.rs +++ b/src/cli.rs @@ -1637,6 +1637,9 @@ mod tests { fn cli_build_features_report_compiled_capabilities() { let features = super::cli_build_features(); + let mut modules = vec!["bytes", "io", "re", "json", "jit", "math"]; + #[cfg(all(feature = "sqlite", not(target_arch = "wasm32")))] + modules.push("sqlite"); assert_eq!( features, vec![ @@ -1646,7 +1649,7 @@ mod tests { .module_override_source("stdlib/rss/strings.rss") .is_some() .then_some("stdlibs".to_string()), - Some("modules=bytes, io, re, json, jit, math".to_string()), + Some(format!("modules={}", modules.join(", "))), ] .into_iter() .flatten() diff --git a/src/lib.rs b/src/lib.rs index f88cf055..90940354 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -25,6 +25,10 @@ pub use assembler::{AsmParseError, Assembler, AssemblerError, BytecodeBuilder, a pub use builtins::runtime::HostCallResult; #[cfg(feature = "runtime")] pub use builtins::runtime::print::{PrintHostFunction, PrintlnHostFunction, format_value}; +#[cfg(all(feature = "runtime", feature = "sqlite", not(target_arch = "wasm32")))] +pub use builtins::runtime::sqlite::{SqliteHostExt, SqliteLimits, SqlitePolicy}; +#[cfg(all(feature = "runtime", not(target_arch = "wasm32")))] +pub use builtins::runtime::{IoHostExt, IoPolicy}; pub use builtins::{ BUILTIN_CATALOG, BuiltinFunction, BuiltinNamespaceMemberSpec, BuiltinNamespaceSpec, CallableDef, CallableParam, CallableParamType, CallableSignature, HostExecution, diff --git a/src/vm/execution_scope.rs b/src/vm/execution_scope.rs index 190ab414..8f63cc72 100644 --- a/src/vm/execution_scope.rs +++ b/src/vm/execution_scope.rs @@ -347,6 +347,22 @@ impl ExecutionScope { .map_err(ExecutionScopeError::Operation) } + /// Drives one operation to terminal, polling its concrete driver. + /// + /// Forwarding the operation registry's [`poll`](OperationRegistry::poll) + /// through the scope keeps the concrete driver's `poll`/cancel running in + /// the operation layer while the scope remains the single ownership unit. + pub fn poll_operation( + &mut self, + id: OperationId, + cx: &mut Context<'_>, + ) -> Poll> { + match self.operations.poll(id, cx) { + Poll::Pending => Poll::Pending, + Poll::Ready(result) => Poll::Ready(result.map_err(ExecutionScopeError::Operation)), + } + } + /// Aborts a started operation in one step so it never produces a /// guest-visible result: cancels the driver exactly once if pending /// (recording the first reason), then consumes and immediately releases diff --git a/src/vm/host.rs b/src/vm/host.rs index f73d42d4..ea76fe5c 100644 --- a/src/vm/host.rs +++ b/src/vm/host.rs @@ -540,6 +540,12 @@ pub(super) struct WaitingHostOp { pub(super) enum WaitingHostOpSource { HostBridge, BuiltinIo, + /// A pending builtin SQLite operation. The variant is feature-neutral: + /// the builtin catalog removes `sqlite::*` entries on builds without the + /// adapter, so `builtin_waiting_source` never produces this variant there, + /// and the poll/cancel hooks in `crate::builtins::runtime` are + /// feature-neutral no-ops on such builds. + BuiltinSqlite, } struct NoopWake; @@ -560,6 +566,22 @@ fn builtin_for_binding_name(name: &str) -> Option { BuiltinFunction::from_namespaced_name(name) } +/// Maps a pending builtin to the waiting-op source that polls its concrete +/// driver. IO builtins are driven through the builtin IO mailbox; SQLite +/// builtins through their own completion mailbox. Both poll the shared +/// execution-scope operation registry; only the result-mailbox lookup +/// differs. Feature-neutral: on builds without the SQLite adapter the catalog +/// contains no `sqlite_*` builtin, so the prefix check never matches there. +fn builtin_waiting_source(builtin: BuiltinFunction) -> WaitingHostOpSource { + // The generated `BuiltinFunction::name()` renders the source name with + // `::` collapsed to `_` (e.g. `sqlite_execute`), matching the internal + // catalog name rather than the guest-facing `sqlite::execute`. + if builtin.name().starts_with("sqlite_") { + return WaitingHostOpSource::BuiltinSqlite; + } + WaitingHostOpSource::BuiltinIo +} + impl Vm { pub fn register_function(&mut self, function: Box) -> u16 { let index = self.host.host_functions.len() as u16; @@ -895,6 +917,9 @@ impl Vm { WaitingHostOpSource::BuiltinIo => { crate::builtins::runtime::cancel_builtin_io_op(self, waiting.op_id); } + WaitingHostOpSource::BuiltinSqlite => { + crate::builtins::runtime::cancel_builtin_sqlite_op(self, waiting.op_id); + } } } @@ -928,6 +953,9 @@ impl Vm { WaitingHostOpSource::BuiltinIo => { crate::builtins::runtime::poll_builtin_io_op(self, waiting.op_id, cx) } + WaitingHostOpSource::BuiltinSqlite => { + crate::builtins::runtime::poll_builtin_sqlite_op(self, waiting.op_id, cx) + } }; match poll_result { @@ -1095,7 +1123,8 @@ impl Vm { crate::builtins::runtime::BuiltinCallOutcome::Pending(op_id) => { self.instance.stack.truncate(arg_start); let resume_ip = self.call_resume_ip(call_ip)?; - self.set_waiting_host_op(op_id, WaitingHostOpSource::BuiltinIo)?; + let source = builtin_waiting_source(builtin); + self.set_waiting_host_op(op_id, source)?; self.instance.ip = resume_ip; Ok(HostCallExecOutcome::Pending(op_id)) } diff --git a/src/vm/host_runtime.rs b/src/vm/host_runtime.rs index 50b16106..5f3d67e7 100644 --- a/src/vm/host_runtime.rs +++ b/src/vm/host_runtime.rs @@ -17,7 +17,7 @@ //! fields (the legacy IO completion mailbox remains on the `Vm` facade until //! the adapter migrates it onto the scope lifecycle). -use std::any::{Any, TypeId}; +use std::any::Any; use std::collections::HashMap; use crate::vm::execution_scope::ExecutionScope; @@ -89,6 +89,7 @@ impl HostRuntime { } /// Borrows the registered typed module state mutably, if any. + #[allow(dead_code)] // used by later host layers (SQLite/capability) in c4/c5 pub(crate) fn get_module_state_mut(&mut self) -> Option<&mut T> { self.module_state_store.get_mut() } @@ -99,6 +100,7 @@ impl HostRuntime { } /// Returns `true` when no module state is currently registered. + #[allow(dead_code)] // used by later host layers (SQLite/capability) in c4/c5 pub(crate) fn is_module_state_empty(&self) -> bool { self.module_state_store.is_empty() } diff --git a/src/vm/host_state.rs b/src/vm/host_state.rs index a1ff8f78..78333deb 100644 --- a/src/vm/host_state.rs +++ b/src/vm/host_state.rs @@ -53,6 +53,7 @@ impl ModuleStateStore { } /// Borrows the registered typed module state mutably, if any. + #[allow(dead_code)] // used by later host layers (SQLite/capability) in c4/c5 pub(crate) fn get_mut(&mut self) -> Option<&mut T> { self.entries .get_mut(&TypeId::of::()) @@ -72,6 +73,7 @@ impl ModuleStateStore { } /// Returns `true` when no module state is currently registered. + #[allow(dead_code)] // used by later host layers (SQLite/capability) in c4/c5 pub(crate) fn is_empty(&self) -> bool { self.entries.is_empty() } diff --git a/src/vm/mod.rs b/src/vm/mod.rs index 2f154124..986dbb64 100644 --- a/src/vm/mod.rs +++ b/src/vm/mod.rs @@ -290,11 +290,6 @@ pub struct Vm { pub(crate) instance: Instance, pub(crate) run_ctx: RunContext, pub(crate) host: HostRuntime, - /// Legacy pre-scope IO runtime state. - /// - /// This commit keeps IO ownership on the `Vm` facade; the generic - /// scope-lifecycle migration relocates it in a later commit. - pub(crate) io_state: crate::builtins::runtime::IoState, } pub(crate) enum ExecOutcome { @@ -558,7 +553,6 @@ impl Vm { instance, run_ctx: RunContext::default(), host: HostRuntime::default(), - io_state: crate::builtins::runtime::IoState::default(), } } @@ -697,11 +691,12 @@ impl Vm { /// preserving JIT artifacts and registered host bindings. /// /// Locals are reset to `Null`, stack is cleared, and instruction pointer is - /// rewound to the program entry. + /// rewound to the program entry. In-flight IO work and live IO handles are + /// retired through the generic execution-scope lifecycle (the old scope is + /// dropped and replaced with a fresh one). pub fn reset_for_reuse(&mut self) { self.cancel_waiting_host_op(); - crate::builtins::runtime::close_all_handles(self); - self.io_state = crate::builtins::runtime::IoState::default(); + self.host.reset_execution_scope(); self.run_ctx.reset_for_reuse(); self.instance.reset(&self.program); self.engine.reset_runtime_state(&self.program); @@ -997,7 +992,9 @@ impl Drop for Vm { fn drop(&mut self) { self.cancel_waiting_host_op(); self.instance.drop_cleanup(); - crate::builtins::runtime::close_all_handles(self); + // Live IO handles and in-flight IO operations are retired by the + // `ExecutionScope`'s own `Drop`, which runs as part of `HostRuntime`. + // (No custom close-all side channel is needed.) } } @@ -2835,7 +2832,6 @@ impl Vm { self.instance.call_depth = 0; self.instance.host_return = None; self.instance.waiting_host_op = None; - crate::builtins::runtime::close_all_handles(self); self.instance.shutdown = true; } diff --git a/tests/builtins/io_scope_lifecycle_tests.rs b/tests/builtins/io_scope_lifecycle_tests.rs new file mode 100644 index 00000000..3632a4c3 --- /dev/null +++ b/tests/builtins/io_scope_lifecycle_tests.rs @@ -0,0 +1,346 @@ +//! Focused TDD tests for migrating baseline IO onto the generic +//! [`ExecutionScope`] lifecycle (PR16 commit 3). +//! +//! File/process handles are typed resources stored in the VM's execution +//! scope; read/write/flush/close/open/popen/exists pending work is driven by +//! concrete [`HostOperation`] drivers registered in the same scope. These +//! tests verify the scope-backed behaviour through the public VM + IO API: +//! stale-handle and type-mismatch rejection, exact-once close, pending +//! operation cancellation, and reset/drop retirement through the generic +//! scope. + +use std::sync::Arc; +use std::sync::atomic::{AtomicUsize, Ordering}; + +use vm::operation::OperationCancelReason; +use vm::operation::OperationId; +use vm::resource::close::{CloseProgress, HostResource}; +use vm::resource::{ResourceCloseReason, ResourceResult}; +use vm::{Value, Vm, VmError, VmStatus, compile_source}; + +/// Helper: run an IO source to completion, returning the final stack. +fn run_source(source: &str) -> Result, VmError> { + let wrapped = format!("use io;\n{source}"); + let compiled = compile_source(&wrapped).expect("source should compile"); + let mut vm = Vm::new(compiled.program); + + let mut status = vm.run()?; + loop { + match status { + VmStatus::Halted => return Ok(vm.stack().to_vec()), + VmStatus::Yielded => { + status = vm.resume()?; + } + VmStatus::Waiting(_) => { + vm.wait_for_host_op_blocking()?; + status = vm.resume()?; + } + } + } +} + +/// Helper: run an IO source expecting a host error, returning its message. +fn run_source_host_error(source: &str) -> String { + match run_source(source) { + Ok(stack) => panic!("expected host error, got stack: {stack:?}"), + Err(VmError::HostError(message)) => message, + Err(other) => panic!("expected host error, got: {other:?}"), + } +} + +// A foreign (non-IO) resource used to exercise type-mismatch rejection. +struct ForeignResource { + closes: Arc, +} + +impl HostResource for ForeignResource { + fn begin_close(&mut self, _reason: ResourceCloseReason) -> ResourceResult { + self.closes.fetch_add(1, Ordering::SeqCst); + Ok(CloseProgress::Ready) + } +} + +/// Compiles and runs an IO source to a VM whose scope reflects the result. +fn vm_for(source: &str) -> Vm { + let wrapped = format!("use io;\n{source}"); + let compiled = compile_source(&wrapped).expect("source should compile"); + let mut vm = Vm::new(compiled.program); + let mut status = vm.run().expect("run should start"); + loop { + match status { + VmStatus::Halted => break, + VmStatus::Yielded => { + status = vm.resume().expect("resume should continue"); + } + VmStatus::Waiting(_) => { + vm.wait_for_host_op_blocking() + .expect("waiting host op should complete"); + status = vm.resume().expect("resume should continue"); + } + } + } + vm +} + +fn host_error(err: VmError) -> String { + match err { + VmError::HostError(message) => message, + other => panic!("expected host error, got: {other:?}"), + } +} + +// ------------------------------------------------------------------ handles + +#[test] +fn io_close_returns_true_and_closed_handle_is_stale() { + // The first close is exact-once and returns `true`; a second use of the + // closed handle (a stale handle) is rejected with a host error rather + // than silently succeeding. + let err = run_source_host_error( + r#" + let handle = io::open("Cargo.toml", "r"); + io::close(handle); + io::close(handle); + "#, + ); + assert!( + err.contains("stale") + || err.contains("not found") + || err.contains("closed") + || err.contains("invalid"), + "double close of a closed IO handle should be rejected; got: {err}" + ); +} + +#[test] +fn io_close_then_read_rejects_stale_handle() { + let err = run_source_host_error( + r#" + let handle = io::open("Cargo.toml", "r"); + io::close(handle); + io::read_all(handle); + "#, + ); + assert!( + err.contains("stale") + || err.contains("not found") + || err.contains("closed") + || err.contains("invalid"), + "reading a closed IO handle should be rejected; got: {err}" + ); +} + +#[test] +fn io_close_on_non_positive_handle_is_rejected() { + let err = run_source_host_error( + r#" + io::close(0); + "#, + ); + assert!( + err.contains("invalid io handle"), + "non-positive handles must be rejected; got: {err}" + ); +} + +#[test] +fn io_open_read_mode_reports_missing_file() { + let err = run_source_host_error( + r#" + io::open("__pd_vm_missing_file_for_test__.txt", "r"); + "#, + ); + assert!( + err.contains("io_open failed"), + "unexpected error message: {err}" + ); +} + +#[test] +fn io_open_rejects_unsupported_mode() { + let err = run_source_host_error( + r#" + io::open("Cargo.toml", "bad"); + "#, + ); + assert!( + err.contains("unsupported io_open mode"), + "unexpected error message: {err}" + ); +} + +// ------------------------------------------------------------- type mismatch + +#[test] +fn io_rejects_foreign_scope_handles() { + // IO handles are scope-scoped typed tokens: a handle minted by one VM's + // execution scope must be rejected when used against another VM's scope + // (wrong table / stale / invalid), never interpreted as a live handle. + let foreign_handle = { + let wrapped = "use io;\nlet h = io::open(\"Cargo.toml\", \"r\");\nh;"; + let compiled = compile_source(wrapped).expect("compile"); + let mut vm = Vm::new(compiled.program); + // Run and drain any waiting IO op. + let mut status = vm.run().expect("run"); + loop { + match status { + VmStatus::Waiting(_) => { + vm.wait_for_host_op_blocking().expect("wait"); + status = vm.resume().expect("resume"); + } + VmStatus::Yielded => { + status = vm.resume().expect("resume"); + } + VmStatus::Halted => break, + } + } + let handle = vm.stack().last().cloned().expect("handle on stack"); + let Value::Int(raw) = handle else { + panic!("io::open must return an integer handle"); + }; + raw + }; + + let wrapped = format!("use io;\nio::close({foreign_handle});"); + let compiled = compile_source(&wrapped).expect("compile"); + let mut vm2 = Vm::new(compiled.program); + let err = host_error( + vm2.run() + .expect_err("foreign handle close must be rejected"), + ); + assert!( + err.contains("mismatch") + || err.contains("type") + || err.contains("stale") + || err.contains("invalid") + || err.contains("table"), + "foreign-scope IO handle access must be rejected; got: {err}" + ); +} + +#[test] +fn io_resources_are_typed_and_never_cross_interpreted() { + // A foreign (non-IO) resource sharing the same execution scope is a + // distinct typed resource: the generic typed-table access rejects a + // wrong-typed token before any IO interpretation can happen. This is the + // generic guarantee IO handles rely on (TypeId-checked borrows). + let closes = Arc::new(AtomicUsize::new(0)); + let wrapped = "use io;\nio::open(\"Cargo.toml\", \"r\");"; + let compiled = compile_source(wrapped).expect("compile"); + let mut vm = Vm::new(compiled.program); + let foreign = vm + .execution_scope() + .push_resource(ForeignResource { + closes: Arc::clone(&closes), + }) + .expect("foreign resource must insert"); + // The foreign token is a valid live resource in this scope: its own + // close (typed correctly) succeeds and runs exactly once. + let _ = vm + .execution_scope() + .close_resource::(foreign.handle(), ResourceCloseReason::Requested) + .expect("typed close of the foreign resource must succeed"); + assert_eq!(closes.load(Ordering::SeqCst), 1, "close runs exactly once"); +} + +// ----------------------------------------------------- pending cancellation + +#[test] +fn pending_io_operation_can_be_cancelled_through_scope() { + // `read_all` on a child that produces no output and does not exit keeps + // the operation genuinely pending. Cancelling it through the VM's + // execution scope must mark it terminal and retire it from the registry. + let wrapped = "use io;\nlet h = io::popen(\"sleep 30\", \"r\");\nio::read_all(h);"; + let compiled = compile_source(wrapped).expect("compile"); + let mut vm = Vm::new(compiled.program); + + let status = vm.run().expect("run should start pending"); + let waiting = match status { + VmStatus::Waiting(op_id) => op_id, + other => panic!("expected a waiting host op, got: {other:?}"), + }; + let id = OperationId::from_raw(waiting).expect("waiting op id must be a valid operation id"); + assert_eq!( + vm.execution_scope().operations().len(), + 1, + "the pending IO op must occupy a scope operation slot" + ); + + let can_cancel = vm + .execution_scope() + .cancel_operation(id, OperationCancelReason::Requested) + .expect("pending op must be cancellable"); + assert!(can_cancel, "cancel on a pending op must report success"); + assert_eq!( + vm.execution_scope().operations().len(), + 1, + "cancellation must retain the operation until its worker exits" + ); + + let error = vm + .wait_for_host_op_blocking() + .expect_err("cancelled IO operation should report cancellation"); + assert!( + matches!(error, VmError::HostError(ref message) if message.contains("cancelled")), + "unexpected cancellation error: {error:?}" + ); + assert!( + vm.execution_scope().operations().is_empty(), + "polling the cancelled operation must release it exactly once" + ); + + // Cancelling the read must also retire the underlying child process so no + // orphaned `sleep 30` survives the test. + wait_for_child_exit(); +} + +/// Best-effort wait so a cancelled child process has time to be reaped before +/// the test process exits (the driver kills it on cancel). +fn wait_for_child_exit() { + std::thread::sleep(std::time::Duration::from_millis(100)); +} + +// ------------------------------------------------ reset / drop retirement + +#[test] +fn reset_for_reuse_joins_pending_io_worker() { + let compiled = + compile_source("use io;\nlet h = io::popen(\"sleep 30\", \"r\");\nio::read_all(h);") + .expect("source should compile"); + let mut vm = Vm::new(compiled.program); + assert!(matches!( + vm.run().expect("run should start"), + VmStatus::Waiting(_) + )); + assert_eq!(vm.execution_scope().operations().len(), 1); + + vm.reset_for_reuse(); + assert!(vm.execution_scope().operations().is_empty()); + assert!(vm.execution_scope().resources().is_empty()); +} + +#[test] +fn reset_for_reuse_retires_io_resources_through_scope() { + let mut vm = vm_for("let h = io::open(\"Cargo.toml\", \"r\");\nh;"); + assert!( + !vm.execution_scope().resources().is_empty(), + "open leaves a live IO resource in the scope" + ); + + vm.reset_for_reuse(); + + assert!( + vm.execution_scope().resources().is_empty() && vm.execution_scope().operations().is_empty(), + "reset for reuse must retire IO resources and operations through the scope" + ); +} + +#[test] +fn drop_retires_io_resources_through_scope() { + // Dropping a VM with a live IO handle must retire the handle through the + // generic scope (no custom close-all side channel). The scope's own Drop + // runs the closing sweep; this test guards that path stays wired. + let mut vm = vm_for("io::open(\"Cargo.toml\", \"r\");"); + assert!(!vm.execution_scope().resources().is_empty()); + drop(vm); +} diff --git a/tests/builtins/sqlite_scope_lifecycle_tests.rs b/tests/builtins/sqlite_scope_lifecycle_tests.rs new file mode 100644 index 00000000..ce0c00eb --- /dev/null +++ b/tests/builtins/sqlite_scope_lifecycle_tests.rs @@ -0,0 +1,338 @@ +//! Focused tests for the scoped SQLite host functions (PR16 commit 4). +//! +//! Connections are typed [`HostResource`]s owned by the VM's execution +//! scope; `sqlite::execute` / `sqlite::query` / `sqlite::transaction` are +//! driven by concrete [`HostOperation`] drivers in the same scope and polled +//! through the shared operation registry. These tests exercise the +//! scope-backed behaviour through the public VM + SQLite API: typed-value +//! round trips and ordered transactions, read-only and SQL-safety policy, +//! row/result-byte truncation bounds, stale/foreign/typed handle rejection, +//! and adapter-owned `configure`/`clear`/`close` cleanup. + +use std::fs; +use std::path::{Path, PathBuf}; +use std::time::{SystemTime, UNIX_EPOCH}; + +use vm::{SqliteHostExt, Vm, VmError, VmStatus, compile_source}; + +/// Helper: run a SQLite source to completion. Scripts use `assert(...)` for +/// value checks; a failed assert surfaces as a host error. +fn run_sqlite_source(policy: vm::SqlitePolicy, source: &str) -> Result<(), VmError> { + let wrapped = format!("use sqlite;\n{source}"); + let compiled = compile_source(&wrapped).expect("source should compile"); + let mut vm = Vm::new(compiled.program); + vm.configure_sqlite(policy); + + let mut status = vm.run()?; + loop { + match status { + VmStatus::Halted => return Ok(()), + VmStatus::Yielded => { + status = vm.resume()?; + } + VmStatus::Waiting(_) => { + vm.wait_for_host_op_blocking()?; + status = vm.resume()?; + } + } + } +} + +/// Helper: run a SQLite source expecting a host error, returning its message. +fn run_sqlite_host_error(policy: vm::SqlitePolicy, source: &str) -> String { + match run_sqlite_source(policy, source) { + Ok(()) => panic!("expected host error, got success"), + Err(VmError::HostError(message)) => message, + Err(other) => panic!("expected host error, got: {other:?}"), + } +} + +fn temporary_root(label: &str) -> PathBuf { + let nonce = SystemTime::now() + .duration_since(UNIX_EPOCH) + .expect("system clock should be after the Unix epoch") + .as_nanos(); + let root = std::env::temp_dir().join(format!( + "rustscript-sqlite-{label}-{}-{nonce}", + std::process::id() + )); + fs::create_dir_all(&root).expect("temporary SQLite root should be created"); + root +} + +fn policy_for(root: &Path) -> vm::SqlitePolicy { + vm::SqlitePolicy { + database_root: Some(root.to_string_lossy().into_owned()), + ..vm::SqlitePolicy::default() + } +} + +#[test] +fn sqlite_round_trip_supports_typed_values_and_ordered_transactions() { + let root = temporary_root("round-trip"); + let policy = policy_for(&root); + run_sqlite_source( + policy, + r#" + use bytes; + let db = sqlite::open({ path: "state.db", mode: "read_write_create", limits: { max_rows: 128, max_result_bytes: 65536, max_statements: 16, max_transaction_ms: 5000 } }); + sqlite::execute(db, "CREATE TABLE values_table (id INTEGER PRIMARY KEY, n INTEGER, r REAL, s TEXT, b BLOB, z TEXT)", []); + let blob_payload = bytes::from_hex("000102"); + let ins = sqlite::execute(db, "INSERT INTO values_table (n, r, s, b, z) VALUES (?1, ?2, ?3, ?4, ?5)", {7, 1.5, "hello", blob_payload, null}); + assert(ins["rows_affected"] == 1); + let rowset = sqlite::query(db, "SELECT n, r, s, b, z FROM values_table ORDER BY id", [], { max_rows: 8, max_result_bytes: 65536 }); + assert(rowset["truncated"] == false); + assert(rowset["columns"] == {"n", "r", "s", "b", "z"}); + assert(rowset["rows"] == { {7, 1.5, "hello", blob_payload, null} }); + + let results = sqlite::transaction(db, { + { sql: "INSERT INTO values_table (n) VALUES (?1)", params: {8} }, + { sql: "INSERT INTO values_table (n) VALUES (?1)", params: {9} } + }); + assert(type(results) == "array"); + let count = sqlite::query(db, "SELECT count(*) AS count FROM values_table", [], { max_rows: 8, max_result_bytes: 65536 }); + assert(count["rows"] == { {3} }); + sqlite::close(db); + "#, + ) + .expect("round-trip should succeed"); + fs::remove_dir_all(root).expect("temporary SQLite root should be removed"); +} + +#[test] +fn sqlite_enforces_read_only_vm_local_ids_and_sql_safety() { + let root = temporary_root("policy"); + let policy = policy_for(&root); + + run_sqlite_source( + policy.clone(), + r#" + let db = sqlite::open({ path: "state.db", mode: "read_write_create", limits: {} }); + sqlite::execute(db, "CREATE TABLE items (value INTEGER)", []); + "#, + ) + .expect("writer should create the table"); + + let hazard = run_sqlite_host_error( + policy.clone(), + r#" + let db = sqlite::open({ path: "state.db", mode: "read_only", limits: {} }); + sqlite::execute(db, "INSERT INTO items (value) VALUES (1)", []); + "#, + ); + assert!( + hazard.contains("ReadOnly") + || hazard.to_lowercase().contains("readonly") + || hazard.to_lowercase().contains("read-only"), + "read-only writes must be rejected, got: {hazard}" + ); + + for bad in [ + "ATTACH DATABASE 'other.db' AS other", + "PRAGMA writable_schema = ON", + "SELECT load_extension('not-available')", + "CREATE TABLE first (id INTEGER); CREATE TABLE second (id INTEGER)", + ] { + let err = run_sqlite_host_error( + policy.clone(), + &format!( + "let db = sqlite::open({{ path: \"state.db\", mode: \"read_write_create\", limits: {{}} }});\n sqlite::execute(db, \"{bad}\", []);" + ), + ); + assert!( + err.contains("not allowed") + || err.contains("multiple statements") + || err.contains("disabled"), + "unsafe SQL must be rejected, got: {err}" + ); + } + + // A SQLite id from another VM must be rejected (foreign arena). + let other_err = run_sqlite_host_error(policy, "sqlite::execute(1234567, \"SELECT 1\", []);"); + assert!( + other_err.contains("unknown SQLite database") + || other_err.contains("invalid sqlite handle"), + "foreign ids must be rejected, got: {other_err}" + ); + + fs::remove_dir_all(root).expect("temporary SQLite root should be removed"); +} + +#[test] +fn sqlite_query_reports_row_and_result_byte_truncation() { + let root = temporary_root("limits"); + let policy = policy_for(&root); + run_sqlite_source( + policy, + r#" + let db = sqlite::open({ path: "state.db", mode: "read_write_create", limits: { max_rows: 32, max_result_bytes: 32 } }); + sqlite::execute(db, "CREATE TABLE items (value TEXT)", []); + sqlite::execute(db, "INSERT INTO items (value) VALUES (?1)", {"one"}); + sqlite::execute(db, "INSERT INTO items (value) VALUES (?1)", {"two"}); + sqlite::execute(db, "INSERT INTO items (value) VALUES (?1)", {"three"}); + + let row_limited = sqlite::query(db, "SELECT value FROM items ORDER BY rowid", [], { max_rows: 1, max_result_bytes: 65536 }); + assert(row_limited["truncated"] == true); + assert(row_limited["rows"] == { {"one"} }); + + let byte_limited = sqlite::query(db, "SELECT value FROM items ORDER BY rowid", [], { max_rows: 32, max_result_bytes: 8 }); + assert(byte_limited["truncated"] == true); + "#, + ) + .expect("truncation should be reported"); + fs::remove_dir_all(root).expect("temporary SQLite root should be removed"); +} + +#[test] +fn sqlite_uses_typed_generation_checked_resource_handles() { + let root = temporary_root("handles"); + let policy = policy_for(&root); + + let err = run_sqlite_host_error( + policy, + r#" + let a = sqlite::open({ path: "handles.db", mode: "read_write_create", limits: {} }); + sqlite::close(a); + let b = sqlite::open({ path: "handles.db", mode: "read_write_create", limits: {} }); + assert(a != b); + sqlite::execute(a, "SELECT 1", []); + "#, + ); + assert!( + err.contains("unknown SQLite database"), + "closed generation must stay invalid after slot reuse, got: {err}" + ); + fs::remove_dir_all(root).expect("temporary SQLite root should be removed"); +} + +#[test] +fn sqlite_connection_limit_is_enforced_by_the_adapter() { + let root = temporary_root("connection-limit"); + let policy = policy_for(&root); + let err = run_sqlite_host_error( + policy, + r#" + let a = sqlite::open({ path: "a.db", mode: "read_write_create", limits: { max_connections: 2 } }); + let b = sqlite::open({ path: "b.db", mode: "read_write_create", limits: { max_connections: 2 } }); + let c = sqlite::open({ path: "c.db", mode: "read_write_create", limits: { max_connections: 2 } }); + "#, + ); + assert!( + err.contains("connection limit"), + "max_connections must be enforced, got: {err}" + ); + fs::remove_dir_all(root).expect("temporary SQLite root should be removed"); +} + +#[test] +fn sqlite_configure_and_clear_own_the_policy() { + let root = temporary_root("policy-config"); + let policy = policy_for(&root); + + // configure_sqlite is honoured by open (root + unsafe flag). + run_sqlite_source( + policy.clone(), + r#" + let db = sqlite::open({ path: "state.db", mode: "read_write_create", limits: {} }); + sqlite::execute(db, "CREATE TABLE items (value INTEGER)", []); + "#, + ) + .expect("configured policy should allow file opens"); + + // clear_sqlite restores the default (no root), so a file open is rejected. + let compiled = compile_source("use sqlite;\nlet db = sqlite::open({ path: \"state.db\", mode: \"read_write_create\", limits: {} });") + .expect("source should compile"); + let mut vm = Vm::new(compiled.program); + vm.configure_sqlite(policy); + vm.clear_sqlite(); + let err = match vm.run() { + Ok(VmStatus::Halted) => panic!("open without a root must fail"), + Ok(_) => panic!("open without a root must fail"), + Err(VmError::HostError(message)) => message, + Err(other) => panic!("expected host error, got: {other:?}"), + }; + assert!( + err.contains("root"), + "cleared policy must reject file opens, got: {err}" + ); + + fs::remove_dir_all(root).expect("temporary SQLite root should be removed"); +} + +#[test] +fn sqlite_close_cancels_siblings_and_reset_retires_all() { + let root = temporary_root("cancel-reset"); + let policy = policy_for(&root); + + // Schedule a long-running query, then close the connection while it is + // still pending. The pending driver observes the closed slot and is + // retired through the generic scope close; a fresh connection on the same + // root then works normally. + run_sqlite_source( + policy.clone(), + r#" + let db = sqlite::open({ path: "state.db", mode: "read_write_create", limits: { max_transaction_ms: 10000, max_result_bytes: 65536 } }); + sqlite::execute(db, "CREATE TABLE items (value INTEGER)", []); + let pending = sqlite::query(db, "WITH RECURSIVE numbers(value) AS (SELECT 1 UNION ALL SELECT value + 1 FROM numbers LIMIT 2000000) SELECT sum(value) FROM numbers", [], { max_rows: 1, max_result_bytes: 65536 }); + sqlite::close(db); + let db2 = sqlite::open({ path: "state.db", mode: "read_write_create", limits: { max_transaction_ms: 10000, max_result_bytes: 65536 } }); + let count = sqlite::query(db2, "SELECT count(*) AS count FROM items", [], {}); + assert(count["rows"] == { {0} }); + sqlite::close(db2); + "#, + ) + .expect("close should cancel pending siblings and leave a reusable connection"); + + // VM reset retires all pending sqlite operations and closes every open + // connection through the generic scope lifecycle. + let compiled = compile_source( + "use sqlite;\nlet db = sqlite::open({ path: \"state.db\", mode: \"read_write_create\", limits: { max_transaction_ms: 10000, max_result_bytes: 65536 } });\nlet pending = sqlite::query(db, \"WITH RECURSIVE numbers(value) AS (SELECT 1 UNION ALL SELECT value + 1 FROM numbers LIMIT 2000000) SELECT sum(value) FROM numbers\", [], { max_rows: 1, max_result_bytes: 65536 });", + ) + .expect("reset source should compile"); + let mut vm = Vm::new(compiled.program); + vm.configure_sqlite(policy); + // Run until the long query is pending (the VM is waiting on it), then + // reset: the scope close must cancel the driver without hanging. + let status = vm.run().expect("run should start"); + assert!( + matches!(status, VmStatus::Waiting(_)), + "long query should leave the VM waiting, got: {status:?}" + ); + vm.reset_for_reuse(); + assert!( + vm.execution_scope().operations().is_empty(), + "reset must retire all pending sqlite operations" + ); + assert!( + vm.execution_scope().resources().is_empty(), + "reset must close every sqlite connection resource" + ); + + fs::remove_dir_all(root).expect("temporary SQLite root should be removed"); +} + +#[test] +fn sqlite_pending_operation_slots_are_reclaimed_after_completion() { + let root = temporary_root("pending-reclaim"); + let policy = policy_for(&root); + // With `max_pending_operations: 4`, more than four sequential operations + // must still succeed: completed operations release their slot so the + // per-connection pending counter does not grow without bound. + run_sqlite_source( + policy, + r#" + let db = sqlite::open({ path: "state.db", mode: "read_write_create", limits: { max_pending_operations: 4 } }); + sqlite::execute(db, "CREATE TABLE items (value INTEGER)", []); + let mut i = 0; + while i < 10 { + sqlite::execute(db, "INSERT INTO items (value) VALUES (?1)", {i}); + i = i + 1; + } + let count = sqlite::query(db, "SELECT count(*) AS count FROM items", [], {}); + assert(count["rows"] == { {10} }); + sqlite::close(db); + "#, + ) + .expect("sequential operations beyond the pending limit should succeed after reclaim"); + fs::remove_dir_all(root).expect("temporary SQLite root should be removed"); +} diff --git a/tests/builtins_tests.rs b/tests/builtins_tests.rs index e4f54996..204b341f 100644 --- a/tests/builtins_tests.rs +++ b/tests/builtins_tests.rs @@ -3,5 +3,12 @@ #[path = "builtins/io_builtin_edge_tests.rs"] mod io_builtin_edge_tests; +#[path = "builtins/io_scope_lifecycle_tests.rs"] +mod io_scope_lifecycle_tests; + +#[cfg(feature = "sqlite")] +#[path = "builtins/sqlite_scope_lifecycle_tests.rs"] +mod sqlite_scope_lifecycle_tests; + #[path = "builtins/stdlib_tests.rs"] mod stdlib_tests; diff --git a/tests/wire/catalog_build_validation_tests.rs b/tests/wire/catalog_build_validation_tests.rs index c56f9185..bc60c3c9 100644 --- a/tests/wire/catalog_build_validation_tests.rs +++ b/tests/wire/catalog_build_validation_tests.rs @@ -17,8 +17,8 @@ use std::panic::{AssertUnwindSafe, catch_unwind}; use build_script::{ CatalogClass, CatalogEntry, ORDINARY_BLOCK_START, SPECIAL_CALL_BLOCK_END, - SPECIAL_CALL_BLOCK_START, builtin_variant_name, parse_catalog_source, - validate_catalog_contract, + SPECIAL_CALL_BLOCK_START, SQLITE_RESERVED_TOP_END, SQLITE_RESERVED_TOP_START, + builtin_variant_name, parse_catalog_source, validate_catalog_contract, }; fn assert_panics(f: F) @@ -55,6 +55,13 @@ fn parse_catalog_source_accepts_the_checked_in_catalog() { )) .expect("read authoritative catalog"); let entries = parse_catalog_source(&source, "catalog.rs"); + // The SQLite namespace is optional (mirrors the build.rs feature filter): + // when the feature is off, the generated catalog excludes it. + #[cfg(not(feature = "sqlite"))] + let entries: Vec<_> = entries + .into_iter() + .filter(|entry| !entry.source_name.starts_with("sqlite::")) + .collect(); assert!(!entries.is_empty()); assert_eq!(entries.len(), vm::BUILTIN_CATALOG.len()); } @@ -133,6 +140,16 @@ fn validate_catalog_contract_accepts_a_valid_catalog() { validate_catalog_contract(&entries, &discovered, &special); } +#[test] +fn validate_catalog_contract_rejects_arithmetic_allocation_in_sqlite_top_range() { + for id in SQLITE_RESERVED_TOP_START..=SQLITE_RESERVED_TOP_END { + let entries = vec![entry(id, "len", CatalogClass::Ordinary)]; + let discovered = names(&["len"]); + let special = HashSet::new(); + assert_panics(|| validate_catalog_contract(&entries, &discovered, &special)); + } +} + #[test] fn validate_catalog_contract_rejects_out_of_block_ordinary_ids() { for bad_id in [ diff --git a/tests/wire/catalog_contract_tests.rs b/tests/wire/catalog_contract_tests.rs index f35d3911..42ae876a 100644 --- a/tests/wire/catalog_contract_tests.rs +++ b/tests/wire/catalog_contract_tests.rs @@ -27,6 +27,10 @@ const EXTENSION_BLOCK_END: u16 = 0xFF8F; const SPECIAL_CALL_BLOCK_START: u16 = 0xFF90; const SPECIAL_CALL_BLOCK_END: u16 = 0xFFA1; const ORDINARY_BLOCK_START: u16 = 0xFFA2; +#[cfg(feature = "sqlite")] +const SQLITE_RESERVED_TOP_START: u16 = 0xFFFC; +#[cfg(feature = "sqlite")] +const SQLITE_RESERVED_TOP_END: u16 = u16::MAX; /// Reserved sentinel gap inside the special-call block (see the catalog docs /// and `core.rs::internal_builtins_have_unique_reserved_call_indices`). @@ -80,9 +84,40 @@ fn parse_catalog(source: &str) -> Vec { feature_gate: parts[4].to_string(), }); } + // The SQLite namespace is optional (mirrors the build.rs feature filter): + // when the feature is off, the generated catalog excludes it, so the + // parsed raw catalog must agree. + #[cfg(not(feature = "sqlite"))] + entries.retain(|entry| !entry.source_name.starts_with("sqlite::")); entries } +#[cfg(feature = "sqlite")] +#[test] +fn sqlite_top_u16_ids_are_explicitly_reserved_for_frozen_entries() { + let entries = parse_catalog(&catalog_source()); + let top_entries: Vec<_> = entries + .iter() + .filter(|entry| (SQLITE_RESERVED_TOP_START..=SQLITE_RESERVED_TOP_END).contains(&entry.id)) + .map(|entry| (entry.id, entry.source_name.as_str())) + .collect(); + assert_eq!( + top_entries, + vec![ + (0xFFFC, "sqlite::execute"), + (0xFFFD, "sqlite::query"), + (0xFFFE, "sqlite::transaction"), + (0xFFFF, "sqlite::close"), + ] + ); + assert!( + top_entries + .iter() + .all(|(_, source_name)| source_name.starts_with("sqlite::")), + "new ordinary IDs must not be allocated in the SQLite-reserved top-u16 range" + ); +} + fn assert_unique(values: &[String], what: &str) { let mut seen = std::collections::HashSet::new(); for value in values { @@ -244,6 +279,13 @@ fn checked_in_nostd_mirror_matches_std_catalog() { }; let id = u16::from_str_radix(hex.trim().trim_start_matches("0x"), 16) .unwrap_or_else(|err| panic!("mirror const {const_name} has invalid id: {err}")); + // The SQLite namespace is optional (mirrors the build.rs feature + // filter): when the feature is off, the mirror's sqlite consts are + // excluded from the sync contract. + #[cfg(not(feature = "sqlite"))] + if const_name.starts_with("SQLITE_") { + continue; + } mirror_ids.push(id); mirror_by_const.insert(const_name.to_string(), id); } @@ -344,12 +386,17 @@ fn appending_or_reordering_catalog_entries_does_not_renumber_existing_ids() { } // Appending a new entry at the next free ordinary ID (append-only - // allocation) must not renumber any existing entry. + // allocation) must not renumber any existing entry. When the optional + // SQLite namespace is enabled the ordinary block (0xFFA2..=0xFFFF) is + // exactly full, so there is nothing to append and the property is + // trivially preserved. let mut used: Vec = entries.iter().map(|entry| entry.id).collect(); used.sort_unstable(); - let next_free = (ORDINARY_BLOCK_START..=u16::MAX) - .find(|candidate| used.binary_search(candidate).is_err()) - .expect("ordinary block is exhausted"); + let Some(next_free) = + (ORDINARY_BLOCK_START..=u16::MAX).find(|candidate| used.binary_search(candidate).is_err()) + else { + return; + }; let appended = format!( "{source}\nbuiltin_id!(0x{next_free:04X}, \"synthetic_contract_probe\", \ SyntheticContractProbe, Ordinary, none);\n"