diff --git a/changelog/entries/2026-07-11-commerce-trust-boundary.json b/changelog/entries/2026-07-11-commerce-trust-boundary.json new file mode 100644 index 000000000..08067408a --- /dev/null +++ b/changelog/entries/2026-07-11-commerce-trust-boundary.json @@ -0,0 +1,10 @@ +{ + "id": "2026-07-11-commerce-trust-boundary", + "version": "0.9.4", + "date": "2026-07-11", + "category": "feat", + "title": "Injection-confined ordering: commerce trust boundary", + "summary": "Ordering tools now mechanically refuse free-text ids, external artifact URLs, and URL-bearing ship-to fields, so untrusted content can never steer an order.", + "features": ["fabricate", "security"], + "mcpTools": ["quote_manufacturing", "authorize_spend", "place_order"] +} diff --git a/changelog/entries/2026-07-11-loon-macro-library.json b/changelog/entries/2026-07-11-loon-macro-library.json new file mode 100644 index 000000000..bc2b1fac4 --- /dev/null +++ b/changelog/entries/2026-07-11-loon-macro-library.json @@ -0,0 +1,10 @@ +{ + "id": "2026-07-11-loon-macro-library", + "version": "0.9.4", + "date": "2026-07-11", + "category": "feat", + "title": "Agent macro library: define_loon / call_loon / use_loons", + "summary": "Agents can define reusable parametric loon macros (smoke-tested at definition time), instantiate them with call_loon, and compose them inside any create_cad_loon program.", + "features": ["loon", "macros", "agents"], + "mcpTools": ["define_loon", "call_loon", "list_loons", "create_cad_loon"] +} diff --git a/changelog/entries/2026-07-11-predict-physics.json b/changelog/entries/2026-07-11-predict-physics.json new file mode 100644 index 000000000..58d97f4d1 --- /dev/null +++ b/changelog/entries/2026-07-11-predict-physics.json @@ -0,0 +1,10 @@ +{ + "id": "2026-07-11-predict-physics", + "version": "0.9.4", + "date": "2026-07-11", + "category": "feat", + "title": "predict_physics: two-tier static FEA with honest receipts", + "summary": "Fast voxel FEA (displacement, von Mises stress, compliance) in ~100ms; predict-tier claims are basis=predicted and roll up provisional, verify-tier certifies with the same solver.", + "features": ["physics", "verification", "receipt"], + "mcpTools": ["predict_physics"] +} diff --git a/changelog/entries/2026-07-11-receipt-claim-basis.json b/changelog/entries/2026-07-11-receipt-claim-basis.json new file mode 100644 index 000000000..d25942448 --- /dev/null +++ b/changelog/entries/2026-07-11-receipt-claim-basis.json @@ -0,0 +1,10 @@ +{ + "id": "2026-07-11-receipt-claim-basis", + "version": "0.9.4", + "date": "2026-07-11", + "category": "feat", + "title": "Receipts distinguish predicted, verified, and measured claims", + "summary": "Claims now carry a basis (predicted/verified/measured); receipts passing only on surrogate predictions roll up as provisional, never pass.", + "features": ["receipt", "verification"], + "mcpTools": ["build_receipt", "verify_receipt", "verify_spec"] +} diff --git a/crates/vcad-kernel-topopt/src/analyze.rs b/crates/vcad-kernel-topopt/src/analyze.rs new file mode 100644 index 000000000..73d1b9b36 --- /dev/null +++ b/crates/vcad-kernel-topopt/src/analyze.rs @@ -0,0 +1,378 @@ +//! Standalone static structural analysis on the voxel FE machinery. +//! +//! This is the fast inner loop of the two-tier physics pattern: the same +//! solver family serves both tiers, and **resolution is the fidelity dial**. +//! A coarse grid answers in milliseconds and is honest about being an +//! estimate (`basis: predicted` upstream); a fine grid is the trusted +//! verify pass. Because both tiers share one discretization and solver, +//! "verify" genuinely refines "predict" rather than being a different +//! oracle with different blind spots. + +use crate::domain::Domain; +use crate::fea::FeSystem; +use crate::spec::{Load, Support}; +use serde::{Deserialize, Serialize}; +use vcad_kernel_tessellate::TriangleMesh; + +/// Specification for a static analysis run. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct AnalysisSpec { + /// Voxel count along the longest axis. Clamped to `[2, 256]`. + /// 32 is the fast predict tier; 64–96 is the verify tier. Trilinear + /// hexes lock in bending below ~4 elements through the thinnest + /// section — going coarser than 32 on slender parts is dishonest, not + /// fast (a res-20 cantilever reads 2.2× too stiff). + #[serde(default = "default_resolution")] + pub resolution: usize, + /// Young's modulus in MPa (N/mm²), e.g. 69_000 for 6061 aluminum. + #[serde(default = "default_youngs_modulus")] + pub youngs_modulus_mpa: f64, + /// Poisson's ratio. + #[serde(default = "default_poisson")] + pub poisson: f64, + /// Applied loads (at least one required). Forces in Newtons. + pub loads: Vec, + /// Supports (at least one required). + pub supports: Vec, +} + +fn default_resolution() -> usize { + 32 +} +fn default_youngs_modulus() -> f64 { + 69_000.0 // 6061-T6 aluminum +} +fn default_poisson() -> f64 { + 0.33 +} + +/// Result of a static analysis solve. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct StaticAnalysis { + /// Compliance `fᵀu` in N·mm — the work done by the loads. Lower is + /// stiffer for the same loads. + pub compliance_n_mm: f64, + /// Maximum nodal displacement magnitude in mm. + pub max_displacement_mm: f64, + /// World position of the most-displaced node, mm. + pub max_displacement_at: [f64; 3], + /// Maximum element-centroid von Mises stress in MPa. Voxel FEA smears + /// stress concentrations; treat as an estimate, tighter at higher + /// resolution. + pub max_von_mises_mpa: f64, + /// World position of the centroid of the most-stressed element, mm. + pub max_stress_at: [f64; 3], + /// Voxel grid dimensions used, `[nx, ny, nz]`. + pub grid: [usize; 3], + /// Voxel edge length in mm. + pub voxel_size_mm: f64, + /// Relative residual the PCG solve reached. + pub relative_residual: f64, + /// Whether the solve converged below tolerance. + pub converged: bool, +} + +/// Errors from static analysis. +#[derive(Debug)] +pub enum AnalyzeError { + /// The specification is invalid. + InvalidSpec(String), + /// Boundary conditions or the domain are unusable. + Fe(crate::fea::FeError), +} + +impl std::fmt::Display for AnalyzeError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + AnalyzeError::InvalidSpec(msg) => write!(f, "invalid analysis spec: {msg}"), + AnalyzeError::Fe(e) => write!(f, "{e}"), + } + } +} + +impl std::error::Error for AnalyzeError { + fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { + match self { + AnalyzeError::Fe(e) => Some(e), + _ => None, + } + } +} + +impl From for AnalyzeError { + fn from(e: crate::fea::FeError) -> Self { + AnalyzeError::Fe(e) + } +} + +fn validate(spec: &AnalysisSpec) -> Result<(), AnalyzeError> { + if spec.loads.is_empty() { + return Err(AnalyzeError::InvalidSpec( + "at least one load is required".into(), + )); + } + if spec.supports.is_empty() { + return Err(AnalyzeError::InvalidSpec( + "at least one support is required".into(), + )); + } + if spec.loads.iter().any(|l| l.force.iter().all(|c| *c == 0.0)) { + return Err(AnalyzeError::InvalidSpec("a load has zero force".into())); + } + if !spec.youngs_modulus_mpa.is_finite() || spec.youngs_modulus_mpa <= 0.0 { + return Err(AnalyzeError::InvalidSpec(format!( + "youngs_modulus_mpa must be positive, got {}", + spec.youngs_modulus_mpa + ))); + } + if !(0.0..0.5).contains(&spec.poisson) { + return Err(AnalyzeError::InvalidSpec(format!( + "poisson must be in [0, 0.5), got {}", + spec.poisson + ))); + } + Ok(()) +} + +const SOLVE_TOL: f64 = 1e-8; +const SOLVE_MAX_ITER: usize = 6000; + +/// Run a static solve on a prepared domain. +pub fn analyze(domain: &Domain, spec: &AnalysisSpec) -> Result { + validate(spec)?; + let sys = FeSystem::build(domain, spec.poisson, &spec.loads, &spec.supports)?; + + // Solve at unit Young's modulus; linear elasticity lets us rescale. + let scales = vec![1.0f64; sys.active_elems.len()]; + let mut u = vec![0.0f64; sys.ndof]; + let relres = sys.solve(&scales, &mut u, SOLVE_TOL, SOLVE_MAX_ITER); + let e = spec.youngs_modulus_mpa; + // u_real = u_unit / E; compliance_real = fᵀu / E. + let compliance = sys.f.iter().zip(&u).map(|(a, b)| a * b).sum::() / e; + + // Max nodal displacement. + let mut max_disp = 0.0f64; + let mut max_disp_node = 0usize; + for n in 0..domain.num_nodes() { + let d2 = u[3 * n].powi(2) + u[3 * n + 1].powi(2) + u[3 * n + 2].powi(2); + if d2 > max_disp { + max_disp = d2; + max_disp_node = n; + } + } + let max_displacement_mm = max_disp.sqrt() / e; + + // Element-centroid von Mises stress. At the element center the shape + // derivative of node k along axis a is s_k[a] / (4h) (trilinear hex). + let (c1, c2, g) = { + let nu = spec.poisson; + ( + (1.0 - nu) / ((1.0 + nu) * (1.0 - 2.0 * nu)), + nu / ((1.0 + nu) * (1.0 - 2.0 * nu)), + 1.0 / (2.0 * (1.0 + nu)), + ) + }; + const SIGNS: [[f64; 3]; 8] = [ + [-1.0, -1.0, -1.0], + [1.0, -1.0, -1.0], + [1.0, 1.0, -1.0], + [-1.0, 1.0, -1.0], + [-1.0, -1.0, 1.0], + [1.0, -1.0, 1.0], + [1.0, 1.0, 1.0], + [-1.0, 1.0, 1.0], + ]; + let inv4h = 1.0 / (4.0 * domain.h); + let mut max_vm = 0.0f64; + let mut max_vm_elem = 0u32; + for (ei, dofs) in sys.edofs.iter().enumerate() { + // Strain at centroid (unit-E displacements). + let mut eps = [0.0f64; 6]; // xx, yy, zz, xy, yz, xz (engineering shear) + for (k, s) in SIGNS.iter().enumerate() { + let ux = u[dofs[3 * k] as usize]; + let uy = u[dofs[3 * k + 1] as usize]; + let uz = u[dofs[3 * k + 2] as usize]; + let (dx, dy, dz) = (s[0] * inv4h, s[1] * inv4h, s[2] * inv4h); + eps[0] += dx * ux; + eps[1] += dy * uy; + eps[2] += dz * uz; + eps[3] += dy * ux + dx * uy; + eps[4] += dz * uy + dy * uz; + eps[5] += dz * ux + dx * uz; + } + // Stress (the unit E cancels against 1/E on u, so this is real MPa). + let sx = c1 * eps[0] + c2 * (eps[1] + eps[2]); + let sy = c1 * eps[1] + c2 * (eps[0] + eps[2]); + let sz = c1 * eps[2] + c2 * (eps[0] + eps[1]); + let (txy, tyz, txz) = (g * eps[3], g * eps[4], g * eps[5]); + let vm = (0.5 * ((sx - sy).powi(2) + (sy - sz).powi(2) + (sz - sx).powi(2)) + + 3.0 * (txy * txy + tyz * tyz + txz * txz)) + .sqrt(); + if vm > max_vm { + max_vm = vm; + max_vm_elem = sys.active_elems[ei]; + } + } + + // World positions for the argmax node/element. + let nxp = domain.nx + 1; + let nyp = domain.ny + 1; + let (nix, niy, niz) = ( + max_disp_node % nxp, + (max_disp_node / nxp) % nyp, + max_disp_node / (nxp * nyp), + ); + let e_us = max_vm_elem as usize; + let (eix, eiy, eiz) = ( + e_us % domain.nx, + (e_us / domain.nx) % domain.ny, + e_us / (domain.nx * domain.ny), + ); + let ecenter = [ + domain.origin[0] + (eix as f64 + 0.5) * domain.h, + domain.origin[1] + (eiy as f64 + 0.5) * domain.h, + domain.origin[2] + (eiz as f64 + 0.5) * domain.h, + ]; + + Ok(StaticAnalysis { + compliance_n_mm: compliance, + max_displacement_mm, + max_displacement_at: domain.node_pos(nix, niy, niz), + max_von_mises_mpa: max_vm, + max_stress_at: ecenter, + grid: [domain.nx, domain.ny, domain.nz], + voxel_size_mm: domain.h, + relative_residual: relres, + converged: relres < 1e-6, + }) +} + +/// Analyze an axis-aligned solid box. +pub fn analyze_box( + min: [f64; 3], + max: [f64; 3], + spec: &AnalysisSpec, +) -> Result { + if (0..3).any(|a| !(max[a] - min[a]).is_finite() || max[a] - min[a] <= 0.0) { + return Err(AnalyzeError::InvalidSpec( + "domain box must have positive size on every axis".into(), + )); + } + let domain = Domain::from_bbox(min, max, spec.resolution); + analyze(&domain, spec) +} + +/// Analyze an existing solid via its tessellation (voxelized like +/// [`crate::optimize_mesh`]). +pub fn analyze_mesh( + mesh: &TriangleMesh, + spec: &AnalysisSpec, +) -> Result { + let domain = Domain::from_mesh(mesh, spec.resolution); + analyze(&domain, spec) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::spec::RegionBox; + + fn cantilever_spec(resolution: usize) -> AnalysisSpec { + AnalysisSpec { + resolution, + youngs_modulus_mpa: 69_000.0, + poisson: 0.33, + loads: vec![Load { + region: RegionBox { + min: [80.0, 0.0, 0.0], + max: [80.0, 10.0, 10.0], + }, + force: [0.0, 0.0, -100.0], + }], + supports: vec![Support { + region: RegionBox { + min: [0.0, 0.0, 0.0], + max: [0.0, 10.0, 10.0], + }, + fix: [true, true, true], + }], + } + } + + #[test] + fn cantilever_matches_beam_theory_order() { + // 80×10×10 mm aluminum cantilever, 100 N tip load. + // Euler–Bernoulli: δ = FL³/(3EI), I = bh³/12 = 10·10³/12 ≈ 833.3 mm⁴ + // δ ≈ 100·512000/(3·69000·833.3) ≈ 0.297 mm. + let a = analyze_box([0.0; 3], [80.0, 10.0, 10.0], &cantilever_spec(32)).unwrap(); + assert!(a.converged, "relres {}", a.relative_residual); + assert!( + a.max_displacement_mm > 0.15 && a.max_displacement_mm < 0.6, + "tip deflection {} outside beam-theory ballpark", + a.max_displacement_mm + ); + // Max deflection is at the loaded tip. + assert!(a.max_displacement_at[0] > 70.0); + // Peak stress near the fixed root: σ = Mc/I ≈ 100·80·5/833 ≈ 48 MPa. + assert!( + a.max_von_mises_mpa > 15.0 && a.max_von_mises_mpa < 150.0, + "root stress {} outside ballpark", + a.max_von_mises_mpa + ); + assert!(a.max_stress_at[0] < 20.0, "peak stress not near root"); + assert!(a.compliance_n_mm > 0.0); + } + + #[test] + fn coarse_predicts_fine_within_tolerance() { + // The two-tier contract: the predict tier must land in the same + // ballpark as the verify tier for a smooth problem. Below ~4 + // elements through the thinnest section trilinear hexes lock in + // bending (res 20 on this beam reads 2.2× too stiff) — which is + // why the predict tier default is 32, not lower. + let coarse = analyze_box([0.0; 3], [80.0, 10.0, 10.0], &cantilever_spec(32)).unwrap(); + let fine = analyze_box([0.0; 3], [80.0, 10.0, 10.0], &cantilever_spec(64)).unwrap(); + let rel = (coarse.max_displacement_mm - fine.max_displacement_mm).abs() + / fine.max_displacement_mm; + assert!( + rel < 0.35, + "coarse {} vs fine {} — rel err {}", + coarse.max_displacement_mm, + fine.max_displacement_mm, + rel + ); + } + + #[test] + fn stiffer_material_deflects_less() { + let mut alu = cantilever_spec(16); + let mut steel = cantilever_spec(16); + steel.youngs_modulus_mpa = 200_000.0; + alu.youngs_modulus_mpa = 69_000.0; + let a = analyze_box([0.0; 3], [80.0, 10.0, 10.0], &alu).unwrap(); + let s = analyze_box([0.0; 3], [80.0, 10.0, 10.0], &steel).unwrap(); + let ratio = a.max_displacement_mm / s.max_displacement_mm; + assert!( + (ratio - 200.0 / 69.0).abs() < 0.05, + "displacement should scale inversely with E; ratio {ratio}" + ); + // Stress is E-independent for a displacement-driven-by-force problem. + assert!((a.max_von_mises_mpa - s.max_von_mises_mpa).abs() < 1e-6); + } + + #[test] + fn invalid_specs_rejected() { + let mut s = cantilever_spec(16); + s.loads.clear(); + assert!(matches!( + analyze_box([0.0; 3], [10.0; 3], &s), + Err(AnalyzeError::InvalidSpec(_)) + )); + let mut s = cantilever_spec(16); + s.youngs_modulus_mpa = -1.0; + assert!(matches!( + analyze_box([0.0; 3], [10.0; 3], &s), + Err(AnalyzeError::InvalidSpec(_)) + )); + } +} diff --git a/crates/vcad-kernel-topopt/src/lib.rs b/crates/vcad-kernel-topopt/src/lib.rs index e01c454c9..cb12c0254 100644 --- a/crates/vcad-kernel-topopt/src/lib.rs +++ b/crates/vcad-kernel-topopt/src/lib.rs @@ -48,12 +48,14 @@ //! assert!(result.compliance_history.len() >= 2); //! ``` +mod analyze; mod domain; mod extract; mod fea; mod simp; mod spec; +pub use analyze::{analyze, analyze_box, analyze_mesh, AnalysisSpec, AnalyzeError, StaticAnalysis}; pub use domain::Domain; pub use fea::FeError; pub use spec::{Load, RegionBox, Support, TopoOptSpec}; diff --git a/crates/vcad-kernel-wasm/src/lib.rs b/crates/vcad-kernel-wasm/src/lib.rs index 0d6998c56..eaa854a1e 100644 --- a/crates/vcad-kernel-wasm/src/lib.rs +++ b/crates/vcad-kernel-wasm/src/lib.rs @@ -451,6 +451,101 @@ pub fn topology_optimize_mesh( topopt_response(result) } +/// Result of a static structural analysis solve (see +/// `vcad_kernel_topopt::analyze`). Two-tier contract: at coarse resolution +/// this is the fast `predicted` path; the same solver at fine resolution is +/// the `verified` path. +#[derive(Serialize, Deserialize)] +#[cfg_attr(feature = "ts-rs", derive(TS))] +#[cfg_attr(feature = "ts-rs", ts(export, export_to = "generated/"))] +pub struct WasmStaticAnalysis { + /// Compliance `fᵀu` in N·mm (lower = stiffer under these loads). + pub compliance: f64, + /// Maximum nodal displacement magnitude in mm. + #[serde(rename = "maxDisplacementMm")] + pub max_displacement_mm: f64, + /// World position of the most-displaced node, mm. + #[serde(rename = "maxDisplacementAt")] + pub max_displacement_at: [f64; 3], + /// Maximum element-centroid von Mises stress in MPa (voxel estimate). + #[serde(rename = "maxVonMisesMpa")] + pub max_von_mises_mpa: f64, + /// World position of the most-stressed element centroid, mm. + #[serde(rename = "maxStressAt")] + pub max_stress_at: [f64; 3], + /// Voxel grid dimensions `[nx, ny, nz]`. + pub grid: [u32; 3], + /// Voxel edge length in mm. + #[serde(rename = "voxelSizeMm")] + pub voxel_size_mm: f64, + /// Relative residual the PCG solve reached. + #[serde(rename = "relativeResidual")] + pub relative_residual: f64, + /// Whether the solve converged. + pub converged: bool, +} + +fn analysis_response( + a: vcad_kernel::vcad_kernel_topopt::StaticAnalysis, +) -> Result { + let out = WasmStaticAnalysis { + compliance: a.compliance_n_mm, + max_displacement_mm: a.max_displacement_mm, + max_displacement_at: a.max_displacement_at, + max_von_mises_mpa: a.max_von_mises_mpa, + max_stress_at: a.max_stress_at, + grid: [a.grid[0] as u32, a.grid[1] as u32, a.grid[2] as u32], + voxel_size_mm: a.voxel_size_mm, + relative_residual: a.relative_residual, + converged: a.converged, + }; + serde_wasm_bindgen::to_value(&out).map_err(|e| JsError::new(&e.to_string())) +} + +/// Static structural analysis of a box solid. +/// +/// `spec_json` is a serialized `vcad_kernel_topopt::AnalysisSpec` (loads, +/// supports, resolution, youngs_modulus_mpa, poisson). +#[wasm_bindgen(js_name = analyzeStaticsBox)] +#[allow(clippy::too_many_arguments)] +pub fn analyze_statics_box( + spec_json: &str, + min_x: f64, + min_y: f64, + min_z: f64, + max_x: f64, + max_y: f64, + max_z: f64, +) -> Result { + let spec: vcad_kernel::vcad_kernel_topopt::AnalysisSpec = + serde_json::from_str(spec_json).map_err(|e| JsError::new(&format!("bad spec: {e}")))?; + let a = vcad_kernel::vcad_kernel_topopt::analyze_box( + [min_x, min_y, min_z], + [max_x, max_y, max_z], + &spec, + ) + .map_err(|e| JsError::new(&e.to_string()))?; + analysis_response(a) +} + +/// Static structural analysis of an existing (closed) evaluated mesh: the +/// mesh interior is voxelized and solved under the given loads/supports. +#[wasm_bindgen(js_name = analyzeStaticsMesh)] +pub fn analyze_statics_mesh( + spec_json: &str, + positions: &[f32], + indices: &[u32], +) -> Result { + let spec: vcad_kernel::vcad_kernel_topopt::AnalysisSpec = + serde_json::from_str(spec_json).map_err(|e| JsError::new(&format!("bad spec: {e}")))?; + let mut mesh = vcad_kernel_tessellate::TriangleMesh::new(); + mesh.vertices = positions.to_vec(); + mesh.indices = indices.to_vec(); + let a = vcad_kernel::vcad_kernel_topopt::analyze_mesh(&mesh, &spec) + .map_err(|e| JsError::new(&e.to_string()))?; + analysis_response(a) +} + /// A 2D sketch segment (line or arc) for WASM input. #[derive(Serialize, Deserialize)] #[serde(tag = "type")] diff --git a/crates/vcad-receipt/src/lib.rs b/crates/vcad-receipt/src/lib.rs index 827ddfe03..52fe3951e 100644 --- a/crates/vcad-receipt/src/lib.rs +++ b/crates/vcad-receipt/src/lib.rs @@ -48,6 +48,31 @@ pub enum ClaimVerdict { Unverifiable, } +/// How a claim's verdict was produced — the evidentiary weight behind it. +/// +/// A surrogate model and a real solver can check the same claim; the verdict +/// alone does not say which one did. `Predicted` marks fast-path estimates +/// (neural surrogates, analytic approximations) that have not been confirmed +/// by the trusted oracle. A receipt whose passing claims rest on predictions +/// rolls up as [`ReceiptVerdict::Provisional`], never `Pass`. +/// +/// Absent on the wire means [`ClaimBasis::Verified`] — every claim written +/// before this field existed came from a real oracle run. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +#[cfg_attr(feature = "ts-rs", derive(ts_rs::TS))] +#[cfg_attr(feature = "ts-rs", ts(export, export_to = "bindings/"))] +pub enum ClaimBasis { + /// A fast estimate (surrogate model, analytic approximation) that the + /// trusted oracle has not confirmed. Good enough to steer, not to ship. + Predicted, + /// The trusted oracle (solver, DRC engine, rule pack) ran for real. + Verified, + /// Confirmed against the physical world (calipers, scale, spectrum + /// analyzer) — e.g. via `record_measurement`. The strongest basis. + Measured, +} + /// The oracle that checked a claim. #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] #[cfg_attr(feature = "ts-rs", derive(ts_rs::TS))] @@ -157,6 +182,11 @@ pub struct ReceiptClaim { pub oracle: OracleRef, /// The verdict. pub verdict: ClaimVerdict, + /// How the verdict was produced. Absent means [`ClaimBasis::Verified`] + /// (see [`ClaimBasis`] for the back-compat rationale). + #[serde(default, skip_serializing_if = "Option::is_none")] + #[cfg_attr(feature = "ts-rs", ts(optional))] + pub basis: Option, /// The claimed/required value — what the design must meet (a spec bound, /// a rule limit, a declared target). #[serde(default, skip_serializing_if = "Option::is_none")] @@ -188,6 +218,7 @@ impl ReceiptClaim { subject: None, oracle, verdict, + basis: None, predicted: None, measured: None, details: None, @@ -228,6 +259,17 @@ impl ReceiptClaim { c } + /// Mark how this verdict was produced. + pub fn with_basis(mut self, basis: ClaimBasis) -> Self { + self.basis = Some(basis); + self + } + + /// The basis, resolving the wire default: absent means `Verified`. + pub fn effective_basis(&self) -> ClaimBasis { + self.basis.unwrap_or(ClaimBasis::Verified) + } + /// Attach the claimed/required value. pub fn with_predicted(mut self, q: ClaimQuantity) -> Self { self.predicted = Some(q); @@ -271,6 +313,35 @@ pub struct ReceiptSignature { pub signature: String, } +/// Basis-aware fail-closed rollup verdict for a whole receipt. +/// +/// Extends [`ClaimVerdict`] with `Provisional`: the receipt *would* pass, +/// but at least one passing claim rests on a [`ClaimBasis::Predicted`] +/// estimate the trusted oracle has not confirmed. Provisional is never a +/// pass — it is a promissory note, redeemed by re-running the slow oracle. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +#[cfg_attr(feature = "ts-rs", derive(ts_rs::TS))] +#[cfg_attr(feature = "ts-rs", ts(export, export_to = "bindings/"))] +pub enum ReceiptVerdict { + /// Every claim passed on verified or measured basis. + Pass, + /// Every claim passed, but at least one only on predicted basis. + Provisional, + /// At least one claim failed (on any basis — a predicted fail is still + /// a fail: the fast path saying "no" is actionable). + Fail, + /// No evidence, or at least one claim could not be checked. + Unverifiable, +} + +impl Default for ReceiptVerdict { + /// Fail-closed: absence of a computed verdict reads as unverifiable. + fn default() -> Self { + ReceiptVerdict::Unverifiable + } +} + /// Aggregate view of a receipt's claims. #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] #[cfg_attr(feature = "ts-rs", derive(ts_rs::TS))] @@ -284,9 +355,17 @@ pub struct ReceiptSummary { pub failed: u32, /// Claims that could not be verified. pub unverifiable: u32, + /// Claims whose verdict rests on a predicted (surrogate) basis. + #[serde(default)] + pub predicted_basis: u32, /// Fail-closed rollup: `Fail` if anything failed, else `Unverifiable` /// if anything (or everything — zero claims) is unverified, else `Pass`. + /// Basis-blind; see [`ReceiptSummary::verdict`] for the basis-aware view. pub overall: ClaimVerdict, + /// Basis-aware rollup ([`DesignReceipt::verdict`]): like `overall`, but + /// an all-pass receipt leaning on predicted claims reads `Provisional`. + #[serde(default)] + pub verdict: ReceiptVerdict, } /// The unified, versioned verification receipt for a design. @@ -366,24 +445,55 @@ impl DesignReceipt { ClaimVerdict::Pass } + /// Basis-aware fail-closed rollup. + /// + /// Same lattice as [`DesignReceipt::overall`], with one refinement: a + /// receipt that would pass but has any claim on + /// [`ClaimBasis::Predicted`] rolls up as + /// [`ReceiptVerdict::Provisional`]. Predictions can steer a design; only + /// verified or measured evidence can certify one. + pub fn verdict(&self) -> ReceiptVerdict { + match self.overall() { + ClaimVerdict::Fail => ReceiptVerdict::Fail, + ClaimVerdict::Unverifiable => ReceiptVerdict::Unverifiable, + ClaimVerdict::Pass => { + if self + .claims + .iter() + .any(|c| c.effective_basis() == ClaimBasis::Predicted) + { + ReceiptVerdict::Provisional + } else { + ReceiptVerdict::Pass + } + } + } + } + /// Count claims by verdict and compute the rollup. pub fn summary(&self) -> ReceiptSummary { let mut passed = 0u32; let mut failed = 0u32; let mut unverifiable = 0u32; + let mut predicted_basis = 0u32; for c in &self.claims { match c.verdict { ClaimVerdict::Pass => passed += 1, ClaimVerdict::Fail => failed += 1, ClaimVerdict::Unverifiable => unverifiable += 1, } + if c.effective_basis() == ClaimBasis::Predicted { + predicted_basis += 1; + } } ReceiptSummary { total: self.claims.len() as u32, passed, failed, unverifiable, + predicted_basis, overall: self.overall(), + verdict: self.verdict(), } } } @@ -441,6 +551,69 @@ mod tests { assert_eq!(c.details.as_deref(), Some("engine down")); } + #[test] + fn predicted_basis_pass_is_provisional_never_pass() { + let r = DesignReceipt::with_claims(vec![ + ReceiptClaim::pass("a", "mechanical", "stiffness ok", oracle()), + ReceiptClaim::pass("b", "mechanical", "first mode ok", oracle()) + .with_basis(ClaimBasis::Predicted), + ]); + // Basis-blind rollup still reads pass; basis-aware one does not. + assert_eq!(r.overall(), ClaimVerdict::Pass); + assert_eq!(r.verdict(), ReceiptVerdict::Provisional); + let s = r.summary(); + assert_eq!(s.predicted_basis, 1); + assert_eq!(s.verdict, ReceiptVerdict::Provisional); + assert_eq!(s.overall, ClaimVerdict::Pass); + } + + #[test] + fn verified_and_measured_basis_pass_cleanly() { + let r = DesignReceipt::with_claims(vec![ + ReceiptClaim::pass("a", "mechanical", "a", oracle()).with_basis(ClaimBasis::Verified), + ReceiptClaim::pass("b", "mechanical", "b", oracle()).with_basis(ClaimBasis::Measured), + // absent basis defaults to verified + ReceiptClaim::pass("c", "pcb", "c", oracle()), + ]); + assert_eq!(r.verdict(), ReceiptVerdict::Pass); + assert_eq!(r.summary().predicted_basis, 0); + } + + #[test] + fn predicted_fail_and_unverifiable_dominate_provisional() { + let fail = + DesignReceipt::with_claims(vec![ReceiptClaim::fail("a", "mechanical", "a", oracle()) + .with_basis(ClaimBasis::Predicted)]); + assert_eq!(fail.verdict(), ReceiptVerdict::Fail); + + let unv = DesignReceipt::with_claims(vec![ + ReceiptClaim::pass("a", "mechanical", "a", oracle()).with_basis(ClaimBasis::Predicted), + ReceiptClaim::unverifiable("b", "pcb", "b", oracle(), "engine down"), + ]); + assert_eq!(unv.verdict(), ReceiptVerdict::Unverifiable); + + // Empty stays fail-closed on both axes. + assert_eq!(DesignReceipt::new().verdict(), ReceiptVerdict::Unverifiable); + assert_eq!(ReceiptVerdict::default(), ReceiptVerdict::Unverifiable); + } + + #[test] + fn basis_wire_form_and_back_compat() { + let c = + ReceiptClaim::pass("a", "mechanical", "a", oracle()).with_basis(ClaimBasis::Predicted); + let json = serde_json::to_value(&c).unwrap(); + assert_eq!(json["basis"], "predicted"); + + // Pre-basis wire shape (no field) parses and reads as verified. + let legacy: ReceiptClaim = serde_json::from_value(serde_json::json!({ + "id": "a", "domain": "pcb", "description": "d", + "oracle": {"id": "o", "version": "1"}, "verdict": "pass" + })) + .unwrap(); + assert_eq!(legacy.basis, None); + assert_eq!(legacy.effective_basis(), ClaimBasis::Verified); + } + #[test] fn wire_shape_round_trips() { let receipt = DesignReceipt { @@ -514,6 +687,8 @@ mod ts_tests { DesignReceipt::export_all().expect("DesignReceipt export failed"); ReceiptClaim::export_all().expect("ReceiptClaim export failed"); ClaimVerdict::export_all().expect("ClaimVerdict export failed"); + ClaimBasis::export_all().expect("ClaimBasis export failed"); + ReceiptVerdict::export_all().expect("ReceiptVerdict export failed"); ClaimQuantity::export_all().expect("ClaimQuantity export failed"); ClaimValue::export_all().expect("ClaimValue export failed"); OracleRef::export_all().expect("OracleRef export failed"); diff --git a/docs/loon-macro-library.md b/docs/loon-macro-library.md new file mode 100644 index 000000000..91690c0fa --- /dev/null +++ b/docs/loon-macro-library.md @@ -0,0 +1,43 @@ +# The loon macro library + +Agents define reusable parametric macros (`define_loon`), instantiate them +(`call_loon`), and compose them inside any program (`create_cad_loon` + +`use_loons` / inline `loons`). A macro is plain loon source — +`[let [fn [params…] …]]` — prepended to programs exactly like the +stdlib. + +## Storage tiers + +| Tier | Mechanism | Survives | +|---|---|---| +| Warm | in-process registry | instance lifetime | +| Local | JSON files under `VCAD_MCP_STATE_DIR/loon-macros` | restarts (stdio/local) | +| Hosted | `mcp_macros` table (migration 036), per-user via `MacroStore` | cold starts, cross-instance | +| Stateless | pass-by-value `macro`/`loons` args | everything (no server state) | + +Hosted tier requires `SUPABASE_URL` + `SUPABASE_SERVICE_ROLE_KEY` and a +signed-in caller; `user_id` is always the verified token subject, never +tool input. Reads hydrate-on-miss (artifact-store pattern); writes are +best-effort and never fail the define. **Migration 036 is written but not +deployed** — run `supabase db push --dry-run` then `supabase db push`. + +## The trust ladder + +1. **Smoke-tested** (shipped): `define_loon` refuses source that doesn't + compile or whose example call yields no geometry. +2. **Certified** (next rung, `certify_loon` — designed, not built): run + verify-tier oracles over the macro's parameter range and store a + `DesignReceipt` with the macro (the `receipt` column in migration 036 + reserves the slot). Sketch: + - Sample the parameter box (corners + centroid, or user-declared ranges + on each param). + - For each sample: instantiate, then run declared claims — + `predict_physics` at `fidelity=verify` (structural limits), + `verify_spec` (geometric spec), `inspect_cad` bounds (mass/volume). + - All samples pass on `basis=verified` → receipt stored, macro shows + `certified: true` in `list_loons`; any fail/unverifiable → fail-closed, + no badge. + - A certified macro's receipt composes: a document built from certified + macros can cite their receipts as claims with `subject: + macro:@` — re-verified (Holds/Stale/Violated) when the + kernel version changes. diff --git a/docs/trust-boundary.md b/docs/trust-boundary.md new file mode 100644 index 000000000..3eaf146a8 --- /dev/null +++ b/docs/trust-boundary.md @@ -0,0 +1,66 @@ +# The commerce trust boundary + +vcad agents do two things that must never touch: they **ingest untrusted +content** (imported STEP/KiCad/Eagle files, part descriptions, datasheets, +supplier listings), and they **hold spend authority** (`authorize_spend`, +`place_order`). A poisoned part description that talks an agent into an +ordering decision is the canonical prompt-injection loss for an agent-facing +CAD tool. This document is the contract that makes that loss structurally +impossible — enforced mechanically in the MCP server, not by trusting the +model. + +## The confinement rules + +Enforced by `packages/mcp/src/trust-boundary.ts` at the single dispatch +choke-point in `server.ts`, before any handler runs. Fail-closed; refusals +carry a stable `TRUST_BOUNDARY:` prefix. CI proof: +`packages/mcp/src/__tests__/trust-boundary.test.ts`. + +### 1. Money-plane tools accept opaque ids only + +`authorize_spend` and `place_order` operate exclusively on ids minted by +vcad tools (`order_id`, `authorization_id`, `idempotency_key`, +`document_id`), restricted to `[A-Za-z0-9._:-]`, ≤128 chars. Free text — a +"part number" from a datasheet, a URL, prose — is refused before the handler +sees it. Parts reach orders only through the resolution pipeline +(`resolve_part` → catalog `family_id`), never as strings. + +### 2. Artifact references are store-scoped + +`fab_artifact_id` binds fab files to an order by reference. Accepted forms: +a bare `art_…` id, a relative `/artifacts/[/]` path (no dot +segments), or an absolute URL on an allowlisted vcad host. An external URL +planted in imported content can never be bound to an order — and the bytes +only ever come from the artifact store by id; the server never fetches a +caller-supplied URL. + +### 3. Fab-bound free text stays plain + +`ship_to`, `material`, and `finish` travel to the fabricator. They must be +plain bounded text: no URLs, no control characters, flat scalar fields, +length-capped. An address is not a place for instructions. + +## What already stood (and this layer completes) + +The money plane was designed with hard gates before this boundary existed: + +- **Human approval is out-of-band.** `authorize_spend` only ever creates a + `pending_human` authorization; the *only* path to `authorized` is a human + on `vcad.io/authorize/`. No MCP tool can approve spend. +- **`doc_hash` gate** — `place_order` re-hashes the document against the + quote; any drift kills the order. +- **Receipt gate** — durable claims re-verify at order time, fail-closed; + a violated receipt refuses the order. +- **Debit chokepoint** — idempotent, capped, single-use authorizations. + +Those gates verify *the design and the money*. The trust boundary verifies +*the provenance of the words* — closing the remaining channel where +untrusted content could steer what gets ordered, where it ships, or which +files get fabricated. + +## Extending the boundary + +When adding a commerce-plane tool (anything that spends, ships, or binds +artifacts to money): add it to `COMMERCE_TOOLS` and classify each argument +as opaque-id / artifact-ref / fab-text in the corresponding table. A +commerce tool with an unclassified free-text argument is a review blocker. diff --git a/packages/engine/src/index.ts b/packages/engine/src/index.ts index edaa39d49..04a84717e 100644 --- a/packages/engine/src/index.ts +++ b/packages/engine/src/index.ts @@ -430,6 +430,60 @@ export interface KernelModule { positions: Float32Array, indices: Uint32Array, ) => unknown; + /** Static structural analysis of a box solid. */ + analyzeStaticsBox?: ( + specJson: string, + minX: number, + minY: number, + minZ: number, + maxX: number, + maxY: number, + maxZ: number, + ) => unknown; + /** Static structural analysis inside a closed evaluated mesh. */ + analyzeStaticsMesh?: ( + specJson: string, + positions: Float32Array, + indices: Uint32Array, + ) => unknown; +} + +/** + * Static analysis parameters (mirrors `vcad_kernel_topopt::AnalysisSpec`; + * unset fields take the kernel defaults). Resolution is the fidelity dial: + * 32 is the fast predict tier, 64–96 the verify tier. + */ +export interface StaticAnalysisSpec { + loads: TopoOptLoad[]; + supports: TopoOptSupport[]; + /** Voxels along the longest domain axis. Default 32. */ + resolution?: number; + /** Young's modulus in MPa (N/mm²). Default 69 000 (6061 aluminum). */ + youngs_modulus_mpa?: number; + /** Poisson's ratio. Default 0.33. */ + poisson?: number; +} + +/** Result of a static analysis solve (mirrors `WasmStaticAnalysis`). */ +export interface StaticAnalysisResult { + /** Compliance `fᵀu` in N·mm (lower = stiffer under these loads). */ + compliance: number; + /** Maximum nodal displacement magnitude in mm. */ + maxDisplacementMm: number; + /** World position of the most-displaced node, mm. */ + maxDisplacementAt: [number, number, number]; + /** Max element-centroid von Mises stress in MPa (voxel estimate). */ + maxVonMisesMpa: number; + /** World position of the most-stressed element centroid, mm. */ + maxStressAt: [number, number, number]; + /** Voxel grid dimensions `[nx, ny, nz]`. */ + grid: [number, number, number]; + /** Voxel edge length in mm. */ + voxelSizeMm: number; + /** Relative residual the PCG solve reached. */ + relativeResidual: number; + /** Whether the solve converged. */ + converged: boolean; } /** Axis-aligned box region (mm) selecting grid nodes for loads/supports. */ @@ -749,6 +803,8 @@ export class Engine { mesh_clearance: (wasmModule as Record).mesh_clearance as KernelModule["mesh_clearance"], topologyOptimizeBox: (wasmModule as Record).topologyOptimizeBox as KernelModule["topologyOptimizeBox"], topologyOptimizeMesh: (wasmModule as Record).topologyOptimizeMesh as KernelModule["topologyOptimizeMesh"], + analyzeStaticsBox: (wasmModule as Record).analyzeStaticsBox as KernelModule["analyzeStaticsBox"], + analyzeStaticsMesh: (wasmModule as Record).analyzeStaticsMesh as KernelModule["analyzeStaticsMesh"], }, compiledWasmModule); } @@ -1029,6 +1085,55 @@ export class Engine { return fn(JSON.stringify(spec), mesh.positions, mesh.indices) as TopoOptResult; } + /** + * Static structural analysis of a solid box under the spec's loads and + * supports. Resolution is the fidelity dial: ~32 answers fast (the + * `predicted` tier), 64–96 is the trusted `verified` tier — same solver, + * finer grid. + */ + analyzeStaticsBox( + min: [number, number, number], + max: [number, number, number], + spec: StaticAnalysisSpec, + ): StaticAnalysisResult { + const fn = this.kernel.analyzeStaticsBox; + if (typeof fn !== "function") { + throw new Error( + "analyzeStaticsBox is not exported by this kernel WASM build — rebuild packages/kernel-wasm", + ); + } + return fn( + JSON.stringify(spec), + min[0], + min[1], + min[2], + max[0], + max[1], + max[2], + ) as StaticAnalysisResult; + } + + /** + * Static structural analysis inside a closed evaluated mesh: the mesh + * interior is voxelized and solved under the given loads/supports. + */ + analyzeStaticsMesh( + mesh: TriangleMesh, + spec: StaticAnalysisSpec, + ): StaticAnalysisResult { + const fn = this.kernel.analyzeStaticsMesh; + if (typeof fn !== "function") { + throw new Error( + "analyzeStaticsMesh is not exported by this kernel WASM build — rebuild packages/kernel-wasm", + ); + } + return fn( + JSON.stringify(spec), + mesh.positions, + mesh.indices, + ) as StaticAnalysisResult; + } + /** Import solids from a STEP file buffer. * * Returns an array of triangle meshes, one for each body in the STEP file. diff --git a/packages/ir/src/generated.ts b/packages/ir/src/generated.ts index 573ac0050..b7243cef0 100644 --- a/packages/ir/src/generated.ts +++ b/packages/ir/src/generated.ts @@ -211,6 +211,20 @@ c: [number, number, number], */ periodic?: [boolean, boolean, boolean], }; +/** + * How a claim's verdict was produced — the evidentiary weight behind it. + * + * A surrogate model and a real solver can check the same claim; the verdict + * alone does not say which one did. `Predicted` marks fast-path estimates + * (neural surrogates, analytic approximations) that have not been confirmed + * by the trusted oracle. A receipt whose passing claims rest on predictions + * rolls up as [`ReceiptVerdict::Provisional`], never `Pass`. + * + * Absent on the wire means [`ClaimBasis::Verified`] — every claim written + * before this field existed came from a real oracle run. + */ +export type ClaimBasis = "predicted" | "verified" | "measured"; + /** * A value with an explicit unit. */ @@ -2385,6 +2399,11 @@ oracle: OracleRef, * The verdict. */ verdict: ClaimVerdict, +/** + * How the verdict was produced. Absent means [`ClaimBasis::Verified`] + * (see [`ClaimBasis`] for the back-compat rationale). + */ +basis?: ClaimBasis, /** * The claimed/required value — what the design must meet (a spec bound, * a rule limit, a declared target). @@ -2445,11 +2464,31 @@ failed: number, * Claims that could not be verified. */ unverifiable: number, +/** + * Claims whose verdict rests on a predicted (surrogate) basis. + */ +predicted_basis: number, /** * Fail-closed rollup: `Fail` if anything failed, else `Unverifiable` * if anything (or everything — zero claims) is unverified, else `Pass`. + * Basis-blind; see [`ReceiptSummary::verdict`] for the basis-aware view. + */ +overall: ClaimVerdict, +/** + * Basis-aware rollup ([`DesignReceipt::verdict`]): like `overall`, but + * an all-pass receipt leaning on predicted claims reads `Provisional`. + */ +verdict: ReceiptVerdict, }; + +/** + * Basis-aware fail-closed rollup verdict for a whole receipt. + * + * Extends [`ClaimVerdict`] with `Provisional`: the receipt *would* pass, + * but at least one passing claim rests on a [`ClaimBasis::Predicted`] + * estimate the trusted oracle has not confirmed. Provisional is never a + * pass — it is a promissory note, redeemed by re-running the slow oracle. */ -overall: ClaimVerdict, }; +export type ReceiptVerdict = "pass" | "provisional" | "fail" | "unverifiable"; /** * Count of violations of one rule. diff --git a/packages/mcp/src/__tests__/loon-macros.test.ts b/packages/mcp/src/__tests__/loon-macros.test.ts new file mode 100644 index 000000000..40fff007b --- /dev/null +++ b/packages/mcp/src/__tests__/loon-macros.test.ts @@ -0,0 +1,227 @@ +/** + * Agent macro library: define → list → call → compose. + * The contract under test: only macros whose smoke call yields geometry + * enter the library, and stored macros compose into arbitrary programs + * exactly like the stdlib. + */ +import { beforeAll, beforeEach, describe, expect, it } from "vitest"; +import { Engine } from "@vcad/engine"; +import { + callLoonTool, + clearMacrosForTest, + defineLoonTool, + listLoonsTool, + setMacroStoreFactoryForTest, + type LoonMacro, +} from "../tools/loon-macros.js"; +import { createCadLoon } from "../tools/loon.js"; +import type { MacroStore } from "../macro-store.js"; +import type { AuthUser } from "../oauth.js"; + +/** In-memory MacroStore double — the durable tier without Supabase. */ +class FakeMacroStore implements MacroStore { + rows = new Map(); + saves = 0; + async load(name: string): Promise { + return this.rows.get(name) ?? null; + } + async list(): Promise { + return [...this.rows.values()]; + } + async save(m: LoonMacro): Promise { + this.saves++; + this.rows.set(m.name, m); + } +} + +const USER: AuthUser = { sub: "user-1", email: "cam@example.com" }; + +let engine: Engine; + +beforeAll(async () => { + engine = await Engine.init(); +}); + +beforeEach(() => { + clearMacrosForTest(); + setMacroStoreFactoryForTest(() => null); +}); + +const FLANGE = { + name: "test-flange", + description: "Disc with a centered bore", + params: [ + { name: "od", example: 40, unit: "mm" }, + { name: "bore", example: 8, unit: "mm" }, + { name: "t", example: 5, unit: "mm" }, + ], + source: + "[let test-flange [fn [od bore t]\n" + + " [pipe [cylinder [/ od 2] t]\n" + + " [difference [cylinder [/ bore 2] [+ t 2]]]]]]", +}; + +const parse = (r: { content: Array<{ text: string }> }) => + JSON.parse(r.content[0].text); + +describe("define_loon", () => { + it("stores a macro that passes its smoke call", async () => { + const out = parse(await defineLoonTool(FLANGE, engine)); + expect(out.name).toBe("test-flange"); + expect(out.version).toBe(1); + expect(out.smoke_call).toBe("[test-flange 40 8 5]"); + }); + + it("refuses source that does not evaluate — nothing enters the library", async () => { + await expect( + defineLoonTool( + { ...FLANGE, source: "[let test-flange [fn [od bore t] [cyllinder od t]]]" }, + engine, + ), + ).rejects.toThrow(/NOT stored/); + expect(parse(await listLoonsTool()).count).toBe(0); + }); + + it("refuses stdlib shadowing and bad names", async () => { + await expect(defineLoonTool({ ...FLANGE, name: "cube" }, engine)).rejects.toThrow(/shadows/); + await expect(defineLoonTool({ ...FLANGE, name: "Bad Name" }, engine)).rejects.toThrow( + /kebab-case/, + ); + }); + + it("redefinition bumps the version", async () => { + await defineLoonTool(FLANGE, engine); + const v2 = parse(await defineLoonTool(FLANGE, engine)); + expect(v2.version).toBe(2); + }); +}); + +describe("call_loon", () => { + it("instantiates with positional args and mints a session", async () => { + await defineLoonTool(FLANGE, engine); + const out = parse( + await callLoonTool( + { name: "test-flange", args: [60, 10, 6], material: "steel" }, + engine, + ), + ); + expect(out.document_id).toBeTruthy(); + expect(out.macro).toBe("test-flange"); + expect(out.document).toContain("steel"); + }); + + it("enforces arity with the declared parameter names", async () => { + await defineLoonTool(FLANGE, engine); + await expect( + callLoonTool({ name: "test-flange", args: [60] }, engine), + ).rejects.toThrow(/takes 3 args \(od, bore, t\)/); + }); + + it("unknown macro lists what exists", async () => { + await defineLoonTool(FLANGE, engine); + await expect(callLoonTool({ name: "nope", args: [] }, engine)).rejects.toThrow( + /defined macros: test-flange/, + ); + }); +}); + +describe("composition via use_loons", () => { + it("stored macros are callable inside create_cad_loon programs", async () => { + await defineLoonTool(FLANGE, engine); + const result = createCadLoon( + { + source: + "[root [union [translate 50 0 0 [test-flange 30 6 4]] [test-flange 40 8 5]] \"aluminum\"]", + use_loons: ["test-flange"], + format: "json", + }, + engine, + ); + const doc = JSON.parse(result.content[0].text); + expect(doc.roots?.length).toBeGreaterThan(0); + }); + + it("STATELESS: inline `loons` work with an empty registry (cold start)", async () => { + // No define_loon — simulates a fresh serverless instance. The macro + // record travels by value, as returned by define_loon. + const result = createCadLoon( + { + source: "[root [test-flange 40 8 5] \"aluminum\"]", + loons: [{ name: FLANGE.name, source: FLANGE.source }], + format: "json", + }, + engine, + ); + const doc = JSON.parse(result.content[0].text); + expect(doc.roots?.length).toBeGreaterThan(0); + }); + + it("STATELESS: call_loon with an inline macro, params optional", async () => { + const out = parse( + await callLoonTool( + { + name: "test-flange", + args: [60, 10, 6], + macro: { name: FLANGE.name, source: FLANGE.source }, + }, + engine, + ), + ); + expect(out.document_id).toBeTruthy(); + }); + + it("define_loon returns the portable macro record", async () => { + const out = parse(await defineLoonTool(FLANGE, engine)); + expect(out.macro).toEqual({ + name: FLANGE.name, + source: FLANGE.source, + params: FLANGE.params, + }); + }); + + it("missing macro in use_loons errors clearly", async () => { + expect(() => + createCadLoon({ source: "[root [cube 1 1 1] \"default\"]", use_loons: ["ghost"] }, engine), + ).toThrow(/unknown loon macro "ghost"/); + }); +}); + +describe("hosted durability (MacroStore)", () => { + it("define_loon saves to the durable store for a signed-in user", async () => { + const store = new FakeMacroStore(); + setMacroStoreFactoryForTest((u) => (u ? store : null)); + await defineLoonTool(FLANGE, engine, USER); + expect(store.saves).toBe(1); + expect(store.rows.get("test-flange")?.version).toBe(1); + }); + + it("cold start: call_loon hydrates the macro from the store on miss", async () => { + const store = new FakeMacroStore(); + store.rows.set("test-flange", { ...FLANGE, version: 3 }); + setMacroStoreFactoryForTest((u) => (u ? store : null)); + // Registry is empty (fresh instance) — the durable row makes the call work. + const out = parse( + await callLoonTool({ name: "test-flange", args: [60, 10, 6] }, engine, USER), + ); + expect(out.document_id).toBeTruthy(); + expect(out.version).toBe(3); + }); + + it("cold start: redefinition continues the cloud version sequence", async () => { + const store = new FakeMacroStore(); + store.rows.set("test-flange", { ...FLANGE, version: 4 }); + setMacroStoreFactoryForTest((u) => (u ? store : null)); + const out = parse(await defineLoonTool(FLANGE, engine, USER)); + expect(out.version).toBe(5); + }); + + it("list_loons merges the cloud library; anonymous users stay warm-only", async () => { + const store = new FakeMacroStore(); + store.rows.set("test-flange", { ...FLANGE, version: 2 }); + setMacroStoreFactoryForTest((u) => (u ? store : null)); + expect(parse(await listLoonsTool(null)).count).toBe(0); + const signedIn = parse(await listLoonsTool(USER)); + expect(signedIn.count).toBe(1); + expect(signedIn.macros[0].version).toBe(2); + }); +}); diff --git a/packages/mcp/src/__tests__/predict-physics.test.ts b/packages/mcp/src/__tests__/predict-physics.test.ts new file mode 100644 index 000000000..69619c1e2 --- /dev/null +++ b/packages/mcp/src/__tests__/predict-physics.test.ts @@ -0,0 +1,102 @@ +/** + * predict_physics: two-tier static FEA with basis-tagged receipt claims. + * The contract under test: predict-tier passes are PROVISIONAL, verify-tier + * passes are PASS, and both tiers agree on the physics ballpark. + */ +import { beforeAll, describe, expect, it } from "vitest"; +import { Engine } from "@vcad/engine"; +import { predictPhysicsTool } from "../tools/physics.js"; + +let engine: Engine; + +beforeAll(async () => { + engine = await Engine.init(); +}); + +interface PhysicsOut { + fidelity: string; + basis: string; + solve_ms: number; + analysis: { + max_displacement_mm: number; + max_von_mises_mpa: number; + converged: boolean; + }; + receipt?: { claims: Array> }; + summary?: { verdict: string; predicted_basis: number; overall: string }; + note?: string; +} + +// 80×10×10 mm aluminum cantilever, 100 N tip load, fixed root. +// Beam theory tip deflection ≈ 0.297 mm; root stress ≈ 48 MPa. +const cantilever = (extra: Record) => ({ + domain_box: { min: [0, 0, 0], max: [80, 10, 10] }, + loads: [ + { region: { min: [80, 0, 0], max: [80, 10, 10] }, force: [0, 0, -100] }, + ], + supports: [{ region: { min: [0, 0, 0], max: [0, 10, 10] } }], + ...extra, +}); + +const run = (args: Record): PhysicsOut => + JSON.parse(predictPhysicsTool(args, engine).content[0].text) as PhysicsOut; + +describe("predict_physics", () => { + it("predict tier: fast, ballpark-correct, provisional receipt", () => { + const out = run( + cantilever({ max_displacement_mm: 0.5, max_von_mises_mpa: 100 }), + ); + expect(out.fidelity).toBe("predict"); + expect(out.basis).toBe("predicted"); + expect(out.analysis.converged).toBe(true); + expect(out.analysis.max_displacement_mm).toBeGreaterThan(0.15); + expect(out.analysis.max_displacement_mm).toBeLessThan(0.5); + expect(out.receipt?.claims).toHaveLength(2); + for (const c of out.receipt!.claims) { + expect(c.basis).toBe("predicted"); + expect(c.verdict).toBe("pass"); + } + // The load-bearing assertion: predicted passes are NOT a clean pass. + expect(out.summary?.overall).toBe("pass"); + expect(out.summary?.verdict).toBe("provisional"); + expect(out.summary?.predicted_basis).toBe(2); + }); + + it("verify tier upgrades the same claims to a clean pass", () => { + const out = run( + cantilever({ + fidelity: "verify", + max_displacement_mm: 0.5, + max_von_mises_mpa: 100, + }), + ); + expect(out.basis).toBe("verified"); + expect(out.summary?.verdict).toBe("pass"); + expect(out.summary?.predicted_basis).toBe(0); + }); + + it("both tiers agree on the physics ballpark", () => { + const p = run(cantilever({})); + const v = run(cantilever({ fidelity: "verify" })); + const rel = + Math.abs(p.analysis.max_displacement_mm - v.analysis.max_displacement_mm) / + v.analysis.max_displacement_mm; + expect(rel).toBeLessThan(0.35); + expect(p.note).toContain("No limits asserted"); + }); + + it("violated limit fails the claim on either basis", () => { + const out = run(cantilever({ max_displacement_mm: 0.01 })); + expect(out.receipt?.claims[0].verdict).toBe("fail"); + expect(out.summary?.verdict).toBe("fail"); + }); + + it("rejects malformed problems", () => { + expect(() => run(cantilever({ part: "also-a-part" }))).toThrow( + /exactly one/, + ); + expect(() => + run({ domain_box: { min: [0, 0, 0], max: [10, 10, 10] }, loads: [] }), + ).toThrow(/loads/); + }); +}); diff --git a/packages/mcp/src/__tests__/tool-surface.fixture.json b/packages/mcp/src/__tests__/tool-surface.fixture.json index 5f554058c..55896ef72 100644 --- a/packages/mcp/src/__tests__/tool-surface.fixture.json +++ b/packages/mcp/src/__tests__/tool-surface.fixture.json @@ -1291,6 +1291,35 @@ "type": "string", "description": "Loon source code defining CAD geometry" }, + "use_loons": { + "type": "array", + "items": { + "type": "string" + }, + "description": "Stored macro names (see list_loons) to prepend as a library — their [let [fn ...]] definitions become callable from `source`, exactly like the stdlib. List dependencies before dependents." + }, + "loons": { + "type": "array", + "description": "STATELESS macro library: macros passed by value (the `macro` records define_loon returns: {name, source}). Prepended like use_loons but with no server-side registry dependency — immune to serverless cold starts. Names here also satisfy use_loons.", + "items": { + "type": "object", + "required": [ + "name", + "source" + ], + "properties": { + "name": { + "type": "string" + }, + "source": { + "type": "string" + }, + "params": { + "type": "array" + } + } + } + }, "format": { "type": "string", "enum": [ @@ -1317,6 +1346,144 @@ "openai/toolInvocation/invoked": "Model updated" } }, + { + "name": "define_loon", + "title": "Define Loon Macro", + "description": "Add a reusable parametric macro to the loon library. Provide loon source defining [let [fn [params...] ]] plus parameter docs with example values. The macro is smoke-tested at definition time (must compile and the example call must yield geometry) — only known-good macros enter the library. Once defined, instantiate with call_loon or compose inside any create_cad_loon program via use_loons. Redefining a name bumps its version. Prefer macros over re-writing the same geometry each session.", + "inputSchema": { + "type": "object", + "properties": { + "name": { + "type": "string", + "description": "kebab-case macro name (2–64 chars). The source must define a loon function of this exact name via [let [fn [...] ...]]." + }, + "description": { + "type": "string", + "description": "One sentence: what the macro builds." + }, + "params": { + "type": "array", + "description": "Ordered parameter docs matching the fn's parameter list. `example` values are used for the definition-time smoke call.", + "items": { + "type": "object", + "required": [ + "name", + "example" + ], + "properties": { + "name": { + "type": "string" + }, + "description": { + "type": "string" + }, + "example": { + "type": "number", + "description": "A representative value; the smoke call uses it." + }, + "unit": { + "type": "string", + "description": "e.g. mm, deg" + } + } + } + }, + "source": { + "type": "string", + "description": "Loon source defining [let [fn [] ]]. May include helper lets/types; the whole block is prepended to calling programs, exactly like the stdlib." + } + }, + "required": [ + "name", + "description", + "params", + "source" + ] + }, + "annotations": { + "readOnlyHint": false + } + }, + { + "name": "call_loon", + "title": "Call Loon Macro", + "description": "Instantiate a stored loon macro into a new document: positional numeric args in the macro's declared order (see list_loons). Returns the document and a document_id session.", + "inputSchema": { + "type": "object", + "properties": { + "name": { + "type": "string", + "description": "Macro to instantiate." + }, + "args": { + "type": "array", + "items": { + "type": "number" + }, + "description": "Positional arguments, in the macro's declared order." + }, + "material": { + "type": "string", + "description": "Material for the instantiated part. Default \"default\"." + }, + "macro": { + "type": "object", + "description": "STATELESS alternative: the macro passed by value (the `macro` record define_loon returned: {name, source, params}). Wins over the server-side registry; immune to serverless cold starts.", + "properties": { + "name": { + "type": "string" + }, + "source": { + "type": "string" + }, + "params": { + "type": "array" + } + }, + "required": [ + "name", + "source" + ] + }, + "format": { + "type": "string", + "enum": [ + "vcode", + "json" + ], + "description": "Document output format (default vcode)." + } + }, + "required": [ + "name", + "args" + ] + }, + "annotations": { + "readOnlyHint": false + }, + "_meta": { + "ui": { + "resourceUri": "ui://vcad/viewer" + }, + "ui/resourceUri": "ui://vcad/viewer", + "openai/outputTemplate": "ui://vcad/viewer-openai.html", + "openai/toolInvocation/invoking": "Modeling geometry…", + "openai/toolInvocation/invoked": "Model updated" + } + }, + { + "name": "list_loons", + "title": "List Loon Macros", + "description": "List the stored loon macro library: names, versions, parameter docs with units and example values. Use before call_loon or create_cad_loon with use_loons.", + "inputSchema": { + "type": "object", + "properties": {} + }, + "annotations": { + "readOnlyHint": true + } + }, { "name": "export_cad", "title": "Export CAD", @@ -1691,6 +1858,186 @@ "openai/toolInvocation/invoked": "Model updated" } }, + { + "name": "predict_physics", + "title": "Predict Physics", + "description": "Fast static structural analysis (voxel FEA): max displacement, max von Mises stress, and compliance for a part or box under world-frame loads (N) and supports, in ~100 ms at fidelity=predict. Pass max_displacement_mm / max_von_mises_mpa limits to get receipt claims: predict-tier claims carry basis=predicted and roll up as a PROVISIONAL receipt — use them to iterate on a design cheaply. When the design settles, re-run with fidelity=verify (same solver, fine grid) to upgrade the claims to basis=verified and a certifiable pass. Loads/supports are box regions (mm, Z-up); zero-thickness boxes select a face. Voxel FEA smears stress concentrations — treat stress as an estimate near fillets.", + "inputSchema": { + "type": "object", + "properties": { + "document_id": { + "type": "string", + "description": "Session document. Required with `part`." + }, + "part": { + "type": "string", + "description": "Part id or name to analyze (its evaluated volume is voxelized). Mutually exclusive with `domain_box`." + }, + "domain_box": { + "type": "object", + "required": [ + "min", + "max" + ], + "properties": { + "min": { + "type": "array", + "items": { + "type": "number" + }, + "minItems": 3, + "maxItems": 3, + "description": "Minimum corner [x, y, z] in mm." + }, + "max": { + "type": "array", + "items": { + "type": "number" + }, + "minItems": 3, + "maxItems": 3, + "description": "Maximum corner [x, y, z] in mm." + } + }, + "description": "Analyze a solid axis-aligned box (mm, world frame, Z-up) instead of a part. Mutually exclusive with `part`." + }, + "loads": { + "type": "array", + "minItems": 1, + "description": "Loads: total force vectors (N) distributed over the grid nodes in each world-frame box region. A zero-thickness box selects the nearest plane of nodes.", + "items": { + "type": "object", + "required": [ + "region", + "force" + ], + "properties": { + "region": { + "type": "object", + "required": [ + "min", + "max" + ], + "properties": { + "min": { + "type": "array", + "items": { + "type": "number" + }, + "minItems": 3, + "maxItems": 3, + "description": "Minimum corner [x, y, z] in mm." + }, + "max": { + "type": "array", + "items": { + "type": "number" + }, + "minItems": 3, + "maxItems": 3, + "description": "Maximum corner [x, y, z] in mm." + } + } + }, + "force": { + "type": "array", + "items": { + "type": "number" + }, + "minItems": 3, + "maxItems": 3, + "description": "Total force [fx, fy, fz] in N." + } + } + } + }, + "supports": { + "type": "array", + "minItems": 1, + "description": "Fixed (anchored) regions.", + "items": { + "type": "object", + "required": [ + "region" + ], + "properties": { + "region": { + "type": "object", + "required": [ + "min", + "max" + ], + "properties": { + "min": { + "type": "array", + "items": { + "type": "number" + }, + "minItems": 3, + "maxItems": 3, + "description": "Minimum corner [x, y, z] in mm." + }, + "max": { + "type": "array", + "items": { + "type": "number" + }, + "minItems": 3, + "maxItems": 3, + "description": "Maximum corner [x, y, z] in mm." + } + } + }, + "fix": { + "type": "array", + "items": { + "type": "boolean" + }, + "minItems": 3, + "maxItems": 3, + "description": "Which translations are fixed [x, y, z]; default all true." + } + } + } + }, + "fidelity": { + "type": "string", + "enum": [ + "predict", + "verify" + ], + "description": "`predict` (default): coarse fast solve, claims stamped basis=predicted — good enough to steer a design. `verify`: fine solve with the same oracle, claims stamped basis=verified — good enough to certify. A receipt passing only on predicted claims reads `provisional`, never `pass`." + }, + "resolution": { + "type": "number", + "description": "Override voxels along the longest axis (predict=32, verify=72 by default). Below ~4 elements through the thinnest section, bending results are unreliable." + }, + "youngs_modulus_mpa": { + "type": "number", + "description": "Young's modulus in MPa. Default 69000 (6061 aluminum)." + }, + "poisson": { + "type": "number", + "description": "Poisson's ratio. Default 0.33." + }, + "max_displacement_mm": { + "type": "number", + "description": "Optional limit: assert max displacement ≤ this (claim physics.static.displacement)." + }, + "max_von_mises_mpa": { + "type": "number", + "description": "Optional limit: assert max von Mises stress ≤ this (claim physics.static.stress). E.g. yield/safety-factor." + } + }, + "required": [ + "loads", + "supports" + ] + }, + "annotations": { + "readOnlyHint": true + } + }, { "name": "predict_print", "title": "Predict Print", diff --git a/packages/mcp/src/__tests__/trust-boundary.test.ts b/packages/mcp/src/__tests__/trust-boundary.test.ts new file mode 100644 index 000000000..e9c82fb16 Binary files /dev/null and b/packages/mcp/src/__tests__/trust-boundary.test.ts differ diff --git a/packages/mcp/src/macro-store.ts b/packages/mcp/src/macro-store.ts new file mode 100644 index 000000000..6a8042903 --- /dev/null +++ b/packages/mcp/src/macro-store.ts @@ -0,0 +1,150 @@ +/** + * MacroStore — durable per-user backing for the loon macro library. + * + * Mirrors the SessionStore seam (session-store.ts): a Supabase-backed + * implementation over PostgREST with service-role auth and hard + * `user_id=eq.` scoping, selected when the env + a signed-in user + * are present; null otherwise (the warm registry + local files in + * loon-macros.ts remain the fallback). All operations are best-effort and + * fail soft — a Supabase hiccup never breaks define/call, it only reduces + * durability, and the warm registry stays the source of truth for the + * instance's lifetime. + */ + +import type { AuthUser } from "./oauth.js"; +import type { LoonMacro } from "./tools/loon-macros.js"; + +/** Durable macro storage, scoped to one verified user. */ +export interface MacroStore { + /** Load one macro by name; null on miss. */ + load(name: string): Promise; + /** List the user's whole library. */ + list(): Promise; + /** Upsert a macro (keyed user_id+name server-side). */ + save(macro: LoonMacro): Promise; +} + +interface MacroRow { + name: string; + version: number; + description: string; + params: LoonMacro["params"]; + source: string; +} + +const rowToMacro = (r: MacroRow): LoonMacro => ({ + name: r.name, + version: r.version ?? 1, + description: r.description ?? "", + params: Array.isArray(r.params) ? r.params : [], + source: r.source, +}); + +/** Test seam: swappable fetch, mirroring session-store's sessionFetch. */ +export let macroFetch: typeof fetch = (...args) => fetch(...args); +export function setMacroFetchForTest(f: typeof fetch): void { + macroFetch = f; +} + +class SupabaseMacroStore implements MacroStore { + constructor( + private supabaseUrl: string, + private serviceRoleKey: string, + private userId: string, + ) {} + + private url(query: string): string { + const uid = encodeURIComponent(this.userId); + return `${this.supabaseUrl}/rest/v1/mcp_macros?user_id=eq.${uid}${query}`; + } + + private headers(extra: Record = {}): Record { + return { + apikey: this.serviceRoleKey, + Authorization: `Bearer ${this.serviceRoleKey}`, + "Content-Type": "application/json", + ...extra, + }; + } + + async load(name: string): Promise { + try { + const res = await macroFetch( + this.url( + `&name=eq.${encodeURIComponent(name)}&select=name,version,description,params,source&limit=1`, + ), + { + method: "GET", + headers: this.headers({ Accept: "application/vnd.pgrst.object+json" }), + }, + ); + if (!res.ok) return null; // 406 = zero rows → miss + return rowToMacro((await res.json()) as MacroRow); + } catch (err) { + console.error("[macro-store] load failed:", err); + return null; + } + } + + async list(): Promise { + try { + const res = await macroFetch( + this.url("&select=name,version,description,params,source&order=name.asc"), + { method: "GET", headers: this.headers() }, + ); + if (!res.ok) return []; + return ((await res.json()) as MacroRow[]).map(rowToMacro); + } catch (err) { + console.error("[macro-store] list failed:", err); + return []; + } + } + + async save(m: LoonMacro): Promise { + try { + const body = [ + { + user_id: this.userId, // always the verified caller — never tool input + name: m.name, + version: m.version, + description: m.description, + params: m.params, + source: m.source, + updated_at: new Date().toISOString(), + }, + ]; + const res = await macroFetch( + `${this.supabaseUrl}/rest/v1/mcp_macros?on_conflict=user_id,name`, + { + method: "POST", + headers: this.headers({ + Prefer: "resolution=merge-duplicates,return=minimal", + }), + body: JSON.stringify(body), + }, + ); + if (!res.ok) { + console.error( + "[macro-store] save failed:", + res.status, + await res.text().catch(() => ""), + ); + } + } catch (err) { + console.error("[macro-store] save failed:", err); + } + } +} + +/** + * Store factory, mirroring createSessionStore: Supabase-backed when the env + * and a signed-in user are present, else null (warm registry + local files + * only). Anonymous callers get no cloud library — a macro library is an + * identity-scoped asset, unlike capability-keyed sessions. + */ +export function createMacroStore(user: AuthUser | null): MacroStore | null { + const url = process.env.SUPABASE_URL; + const key = process.env.SUPABASE_SERVICE_ROLE_KEY; + if (!url || !key || !user) return null; + return new SupabaseMacroStore(url, key, user.sub); +} diff --git a/packages/mcp/src/receipt-unified.ts b/packages/mcp/src/receipt-unified.ts index face2f769..9ed1c18fc 100644 --- a/packages/mcp/src/receipt-unified.ts +++ b/packages/mcp/src/receipt-unified.ts @@ -12,6 +12,7 @@ */ import type { + ClaimBasis, ClaimQuantity, ClaimVerdict, DesignReceipt, @@ -19,6 +20,7 @@ import type { Receipt, ReceiptClaim, ReceiptSummary, + ReceiptVerdict, } from "@vcad/ir"; import type { EnclosureFitReport } from "@vcad/engine"; @@ -66,6 +68,28 @@ export function overallVerdict(claims: readonly ReceiptClaim[]): ClaimVerdict { return "pass"; } +/** The claim's basis, resolving the wire default: absent means verified. */ +export function effectiveBasis(c: ReceiptClaim): ClaimBasis { + return c.basis ?? "verified"; +} + +/** + * Basis-aware fail-closed rollup, mirroring `DesignReceipt::verdict` in + * Rust: like {@link overallVerdict}, except an all-pass receipt with any + * claim resting on a `predicted` (surrogate) basis reads `provisional` — + * predictions can steer a design; only verified or measured evidence can + * certify one. + */ +export function receiptVerdict( + claims: readonly ReceiptClaim[], +): ReceiptVerdict { + const overall = overallVerdict(claims); + if (overall !== "pass") return overall; + return claims.some((c) => effectiveBasis(c) === "predicted") + ? "provisional" + : "pass"; +} + /** Counts by verdict plus the rollup, mirroring `DesignReceipt::summary`. */ export function summarize(receipt: DesignReceipt): ReceiptSummary { const count = (v: ClaimVerdict) => @@ -75,7 +99,11 @@ export function summarize(receipt: DesignReceipt): ReceiptSummary { passed: count("pass"), failed: count("fail"), unverifiable: count("unverifiable"), + predicted_basis: receipt.claims.filter( + (c) => effectiveBasis(c) === "predicted", + ).length, overall: overallVerdict(receipt.claims), + verdict: receiptVerdict(receipt.claims), }; } diff --git a/packages/mcp/src/server.ts b/packages/mcp/src/server.ts index 724a83342..dcce6d80d 100644 --- a/packages/mcp/src/server.ts +++ b/packages/mcp/src/server.ts @@ -92,6 +92,7 @@ import { OPENAI_WIDGET_CSP, } from "./viewer.js"; import { fireToolAlert } from "./notify.js"; +import { checkCommerceBoundary } from "./trust-boundary.js"; import { configureTelemetry, flushTelemetry } from "./telemetry.js"; import { artifactStoreInfo as artifactStoreInfoLocal } from "./tools/artifact-store.js"; @@ -128,6 +129,8 @@ import { toolDefs as verifyToolDefs } from "./tools/verify.js"; import { toolDefs as verifySpecToolDefs } from "./tools/verify-spec.js"; import { toolDefs as clearanceToolDefs } from "./tools/clearance.js"; import { toolDefs as topoptToolDefs } from "./tools/topopt.js"; +import { toolDefs as physicsToolDefs } from "./tools/physics.js"; +import { toolDefs as loonMacroToolDefs } from "./tools/loon-macros.js"; import { toolDefs as dfmToolDefs } from "./tools/dfm.js"; import { toolDefs as sheetMetalToolDefs } from "./tools/sheet-metal.js"; import { toolDefs as acousticsToolDefs } from "./tools/acoustics.js"; @@ -303,6 +306,8 @@ const STATIC_TOOL_DEFS: readonly ToolDef[] = [ ...verifySpecToolDefs, ...clearanceToolDefs, ...topoptToolDefs, + ...physicsToolDefs, + ...loonMacroToolDefs, ...dfmToolDefs, ...sheetMetalToolDefs, ...acousticsToolDefs, @@ -367,6 +372,10 @@ const LIST_TOOL_ORDER: readonly string[] = [ "apply_edits", // ── Loon DSL one-shot + core see/measure/export ──────────── "create_cad_loon", + // ── Agent macro library (define once, instantiate anywhere) ── + "define_loon", + "call_loon", + "list_loons", "export_cad", "inspect_cad", "measure", @@ -376,6 +385,8 @@ const LIST_TOOL_ORDER: readonly string[] = [ "parameter_gradient", // ── Topology optimization ────────────────────────────────── "topology_optimize", + // ── Two-tier static physics (predict fast, verify to certify) ── + "predict_physics", // ── Print-then-measure calibration loop (3DP) ────────────── "predict_print", "record_measurement", @@ -1316,6 +1327,21 @@ export async function createServer( return unknownResult; } + // ── Trust boundary (commerce plane) ──────────────────────────── + // Mechanical pre-dispatch guard: money-plane tools take opaque ids + // and store-scoped artifact refs only, so content read from imported + // files or part listings can never steer an order. Fail-closed, + // before the handler ever sees the arguments (docs/trust-boundary.md). + const boundary = checkCommerceBoundary(name, args); + if (!boundary.ok) { + const refusal: ToolResult = { + content: [{ type: "text", text: boundary.reason ?? "TRUST_BOUNDARY: refused" }], + isError: true, + }; + fireToolAlert(name, args, refusal); + return refusal; + } + const result = await def.handler(args, ctx); // ── MCP Apps: attach preview handle for geometry tools ────── diff --git a/packages/mcp/src/tools/loon-macros.ts b/packages/mcp/src/tools/loon-macros.ts new file mode 100644 index 000000000..aef605638 --- /dev/null +++ b/packages/mcp/src/tools/loon-macros.ts @@ -0,0 +1,494 @@ +/** + * Agent tool-making layer: define, list, and call parametric loon macros. + * + * A macro is named loon source — `[let [fn [params…] …]]` — exactly + * the idiom the stdlib itself is written in (lib/src/lib.loon). Definitions + * are prepended to programs the same way the stdlib is, so no engine or + * language change is involved: the macro layer turns vcad from a stateless + * kernel into an accumulating library. + * + * The trust ladder starts at definition time: `define_loon` refuses source + * that does not compile, and refuses a macro whose smoke call (with the + * declared example arguments) does not evaluate to a non-empty scene. What + * enters the library is known-good by construction; receipt-certified + * macros (claims over the parameter range) are the planned next rung. + * + * Storage v1: process-warm registry + file persistence under + * VCAD_MCP_STATE_DIR for local/stdio use. Hosted durability (a per-user + * mcp_macros table) is a follow-up — the MacroStore seam is already shaped + * for it. + */ + +import { existsSync, mkdirSync, readFileSync, readdirSync, writeFileSync } from "node:fs"; +import { join } from "node:path"; +import type { Engine } from "@vcad/engine"; +import { toVCode } from "@vcad/ir"; +import { registerSession } from "./session.js"; +import { resolveWithinRoot } from "./safe-path.js"; +import { behavior, type ToolDef } from "./tool-def.js"; +import { createMacroStore, type MacroStore } from "../macro-store.js"; +import type { AuthUser } from "../oauth.js"; + +/** One stored macro. */ +export interface LoonMacro { + /** kebab-case name; also the loon function it must define. */ + name: string; + /** What the macro builds; shown in list_loons. */ + description: string; + /** Ordered parameter docs (names must match the fn's parameter list). */ + params: Array<{ + name: string; + description?: string; + /** Example value used for the definition-time smoke call. */ + example: number; + unit?: string; + }>; + /** Loon source containing `[let [fn [...] ...]]`. May define + * helpers; everything is prepended together at call time. */ + source: string; + /** Monotone version, bumped on redefinition. */ + version: number; +} + +const MACRO_NAME = /^[a-z][a-z0-9-]{1,63}$/; + +/** Names the stdlib already claims — a macro may not shadow them. */ +const RESERVED = new Set([ + "cube", "cylinder", "sphere", "cone", "torus", "wedge", "prism", + "union", "difference", "intersection", "translate", "rotate", "scale", + "mirror", "extrude", "revolve", "shell", "fillet", "chamfer", + "sweep-line", "sweep-helix", "loft", "loft-closed", "linear-pattern", + "circular-pattern", "sketch", "line", "arc", "root", "pipe", "let", + "fn", "type", "assembly", "part", "instance", +]); + +/** Process-warm registry. Hosted instances keep this for their lifetime; + * local/stdio instances also persist to disk (see load/persist). */ +const registry = new Map(); +let hydrated = false; +let diskEnabled = true; +let storeFactory: (user: AuthUser | null) => MacroStore | null = createMacroStore; + +function macroDir(): string { + return join(process.env.VCAD_MCP_STATE_DIR ?? process.cwd(), "loon-macros"); +} + +function hydrateFromDisk(): void { + if (hydrated) return; + hydrated = true; + const dir = macroDir(); + if (!existsSync(dir)) return; + for (const f of readdirSync(dir)) { + if (!f.endsWith(".json")) continue; + try { + const m = JSON.parse(readFileSync(join(dir, f), "utf8")) as LoonMacro; + if (MACRO_NAME.test(m.name) && typeof m.source === "string") { + registry.set(m.name, m); + } + } catch { + // A corrupt file never blocks the library; it is simply skipped. + } + } +} + +function persistToDisk(m: LoonMacro): void { + if (!diskEnabled) return; + try { + const dir = macroDir(); + mkdirSync(dir, { recursive: true }); + const path = resolveWithinRoot(`${m.name}.json`, dir); + writeFileSync(path, JSON.stringify(m, null, 2)); + } catch { + // Warm registry still holds it; disk persistence is best-effort. + } +} + +/** + * Hydrate-on-miss from the durable per-user store (artifact-store pattern): + * requested names absent from the warm registry are fetched and cached; + * with no names given, the user's whole cloud library is merged in (higher + * version wins). Fail-soft: no store or fetch error just means warm-only. + */ +export async function hydrateMacros( + user: AuthUser | null, + names?: string[], +): Promise { + const store = storeFactory(user); + if (!store) return; + hydrateFromDisk(); + if (names) { + const misses = names.filter((n) => !registry.has(n)); + for (const n of misses) { + const m = await store.load(n); + if (m) registry.set(m.name, m); + } + return; + } + for (const m of await store.list()) { + const warm = registry.get(m.name); + if (!warm || m.version >= warm.version) registry.set(m.name, m); + } +} + +/** An inline (pass-by-value) macro: the stateless alternative to the warm + * registry. `define_loon` returns this exact shape so agents can carry + * macros across instances/sessions without any server state. */ +export interface InlineLoon { + name: string; + source: string; + params?: LoonMacro["params"]; +} + +/** Look up macros: inline definitions win, then the warm registry. */ +export function getMacros( + names: string[], + inline?: InlineLoon[], +): LoonMacro[] { + hydrateFromDisk(); + const byValue = new Map( + (inline ?? []).map((m) => [ + m.name, + { description: "", params: m.params ?? [], version: 0, ...m } as LoonMacro, + ]), + ); + return names.map((n) => { + const m = byValue.get(n) ?? registry.get(n); + if (!m) { + const known = [...registry.keys()].sort().join(", ") || "(none defined)"; + throw new Error( + `unknown loon macro "${n}" — defined macros: ${known}. ` + + `Stateless alternative: pass the macro by value via \`loons\`.`, + ); + } + return m; + }); +} + +/** Concatenated source of the given macros, dependency-blind (macros may + * reference each other; callers list dependencies first). */ +export function macroPrelude(names: string[], inline?: InlineLoon[]): string { + return getMacros(names, inline) + .map((m) => `; macro ${m.name} v${m.version}\n${m.source}`) + .join("\n\n"); +} + +const loonNum = (v: number): string => + Number.isFinite(v) ? String(v) : (() => { throw new Error(`non-finite argument ${v}`); })(); + +/** Compose `[root [ args…] material]` call site. */ +function callSite(m: LoonMacro, args: number[], material: string): string { + const argSrc = args.map(loonNum).join(" "); + return `[root [${m.name} ${argSrc}] "${material}"]`; +} + +// ── define_loon ──────────────────────────────────────────────────────── + +export const defineLoonSchema = { + type: "object" as const, + required: ["name", "description", "params", "source"], + properties: { + name: { + type: "string" as const, + description: + "kebab-case macro name (2–64 chars). The source must define a loon " + + "function of this exact name via [let [fn [...] ...]].", + }, + description: { + type: "string" as const, + description: "One sentence: what the macro builds.", + }, + params: { + type: "array" as const, + description: + "Ordered parameter docs matching the fn's parameter list. `example` " + + "values are used for the definition-time smoke call.", + items: { + type: "object" as const, + required: ["name", "example"], + properties: { + name: { type: "string" as const }, + description: { type: "string" as const }, + example: { + type: "number" as const, + description: "A representative value; the smoke call uses it.", + }, + unit: { type: "string" as const, description: "e.g. mm, deg" }, + }, + }, + }, + source: { + type: "string" as const, + description: + "Loon source defining [let [fn [] ]]. " + + "May include helper lets/types; the whole block is prepended to " + + "calling programs, exactly like the stdlib.", + }, + }, +}; + +interface DefineArgs { + name: string; + description: string; + params: LoonMacro["params"]; + source: string; +} + +export async function defineLoonTool( + args: Record, + engine: Engine, + user: AuthUser | null = null, +): Promise<{ content: Array<{ type: "text"; text: string }> }> { + hydrateFromDisk(); + const a = args as unknown as DefineArgs; + // Pull any cloud copy first so redefinition on a cold instance continues + // the version sequence instead of restarting it. + await hydrateMacros(user, [a.name]).catch(() => {}); + if (!MACRO_NAME.test(a.name)) { + throw new Error( + `define_loon: name must be kebab-case ([a-z][a-z0-9-]{1,63}), got "${a.name}"`, + ); + } + if (RESERVED.has(a.name)) { + throw new Error(`define_loon: "${a.name}" shadows a stdlib name`); + } + if (!a.source.includes(`[let ${a.name} `) && !a.source.includes(`[let ${a.name}\n`)) { + throw new Error( + `define_loon: source must define the macro via [let ${a.name} [fn ...]]`, + ); + } + if (!Array.isArray(a.params)) { + throw new Error("define_loon: params must be an array (may be empty)"); + } + + // Definition-time smoke call: the macro must compile AND its example + // instantiation must evaluate to a non-empty scene. Known-good by + // construction or not in the library. + const candidate: LoonMacro = { + name: a.name, + description: String(a.description ?? ""), + params: a.params, + source: a.source, + version: (registry.get(a.name)?.version ?? 0) + 1, + }; + const examples = candidate.params.map((p) => { + if (typeof p.example !== "number" || !Number.isFinite(p.example)) { + throw new Error(`define_loon: param "${p.name}" needs a finite example value`); + } + return p.example; + }); + const smoke = `${candidate.source}\n\n${callSite(candidate, examples, "default")}`; + let doc; + try { + doc = engine.evalVcadSource(smoke); + } catch (e) { + throw new Error( + `define_loon: smoke call [${a.name} ${examples.join(" ")}] failed to ` + + `evaluate — macro NOT stored. Loon error: ${e instanceof Error ? e.message : e}`, + ); + } + if (!doc || !doc.roots?.length || !Object.keys(doc.nodes ?? {}).length) { + throw new Error( + `define_loon: smoke call produced an empty scene — macro NOT stored`, + ); + } + + registry.set(candidate.name, candidate); + persistToDisk(candidate); + // Durable per-user copy (hosted). Best-effort: a store failure reduces + // durability, never breaks the define. + await storeFactory(user)?.save(candidate).catch(() => {}); + return { + content: [ + { + type: "text", + text: JSON.stringify( + { + name: candidate.name, + version: candidate.version, + smoke_call: `[${candidate.name} ${examples.join(" ")}]`, + verified: "compiles + example instantiation yields geometry", + usage: + `call_loon {name: "${candidate.name}", args: [...]} or ` + + `create_cad_loon with use_loons: ["${candidate.name}"]`, + // Pass-by-value record: carry this across sessions/instances and + // replay via `loons` — no server state required. + macro: { + name: candidate.name, + source: candidate.source, + params: candidate.params, + }, + }, + null, + 2, + ), + }, + ], + }; +} + +// ── call_loon ────────────────────────────────────────────────────────── + +export const callLoonSchema = { + type: "object" as const, + required: ["name", "args"], + properties: { + name: { type: "string" as const, description: "Macro to instantiate." }, + args: { + type: "array" as const, + items: { type: "number" as const }, + description: "Positional arguments, in the macro's declared order.", + }, + material: { + type: "string" as const, + description: "Material for the instantiated part. Default \"default\".", + }, + macro: { + type: "object" as const, + description: + "STATELESS alternative: the macro passed by value (the `macro` " + + "record define_loon returned: {name, source, params}). Wins over " + + "the server-side registry; immune to serverless cold starts.", + properties: { + name: { type: "string" as const }, + source: { type: "string" as const }, + params: { type: "array" as const }, + }, + required: ["name", "source"], + }, + format: { + type: "string" as const, + enum: ["vcode", "json"], + description: "Document output format (default vcode).", + }, + }, +}; + +interface CallArgs { + name: string; + args: number[]; + material?: string; + macro?: InlineLoon; + format?: "vcode" | "json"; +} + +export async function callLoonTool( + args: Record, + engine: Engine, + user: AuthUser | null = null, +): Promise<{ content: Array<{ type: "text"; text: string }> }> { + const a = args as unknown as CallArgs; + if (!a.macro) await hydrateMacros(user, [String(a.name)]).catch(() => {}); + const [m] = getMacros([String(a.name)], a.macro ? [a.macro] : undefined); + if (!Array.isArray(a.args)) throw new Error("call_loon: `args` must be an array"); + // Arity is only checkable when the macro declares params (an inline macro + // may omit them — loon itself then reports any mismatch). + const declaredArity = a.macro && !a.macro.params ? undefined : m.params.length; + if (declaredArity !== undefined && a.args.length !== declaredArity) { + throw new Error( + `call_loon: ${m.name} takes ${declaredArity} args ` + + `(${m.params.map((p) => p.name).join(", ")}), got ${a.args.length}`, + ); + } + const source = `${m.source}\n\n${callSite(m, a.args, a.material ?? "default")}`; + const doc = engine.evalVcadSource(source); + if (!doc) { + throw new Error("call_loon: loon evaluation not supported by this engine build"); + } + const documentId = registerSession(doc); + const text = a.format === "json" ? JSON.stringify(doc, null, 2) : toVCode(doc); + return { + content: [ + { + type: "text", + text: JSON.stringify( + { + document_id: documentId, + macro: m.name, + version: m.version, + document: text, + }, + null, + 2, + ), + }, + ], + }; +} + +// ── list_loons ───────────────────────────────────────────────────────── + +export async function listLoonsTool( + user: AuthUser | null = null, +): Promise<{ content: Array<{ type: "text"; text: string }> }> { + hydrateFromDisk(); + await hydrateMacros(user).catch(() => {}); + const macros = [...registry.values()] + .sort((x, y) => x.name.localeCompare(y.name)) + .map((m) => ({ + name: m.name, + version: m.version, + description: m.description, + params: m.params.map((p) => ({ + name: p.name, + ...(p.unit ? { unit: p.unit } : {}), + ...(p.description ? { description: p.description } : {}), + example: p.example, + })), + })); + return { + content: [ + { type: "text", text: JSON.stringify({ count: macros.length, macros }, null, 2) }, + ], + }; +} + +/** Test seam: empty warm registry, no disk reads or writes. */ +export function clearMacrosForTest(): void { + registry.clear(); + hydrated = true; // skip disk hydration + diskEnabled = false; +} + +/** Test seam: swap the durable-store factory (null = no cloud). */ +export function setMacroStoreFactoryForTest( + f: (user: AuthUser | null) => MacroStore | null, +): void { + storeFactory = f; +} + +export const toolDefs: ToolDef[] = [ + { + name: "define_loon", + pack: null, + description: + "Add a reusable parametric macro to the loon library. Provide loon source defining " + + "[let [fn [params...] ]] plus parameter docs with example values. " + + "The macro is smoke-tested at definition time (must compile and the example call must " + + "yield geometry) — only known-good macros enter the library. Once defined, instantiate " + + "with call_loon or compose inside any create_cad_loon program via use_loons. Redefining " + + "a name bumps its version. Prefer macros over re-writing the same geometry each session.", + inputSchema: defineLoonSchema, + handler: (args, ctx) => defineLoonTool(args, ctx.engine, ctx.user), + behavior: behavior({}), + }, + { + name: "call_loon", + pack: null, + description: + "Instantiate a stored loon macro into a new document: positional numeric args in the " + + "macro's declared order (see list_loons). Returns the document and a document_id session.", + inputSchema: callLoonSchema, + handler: (args, ctx) => callLoonTool(args, ctx.engine, ctx.user), + behavior: behavior({ writesDoc: true, geometry: true, mount: true }), + }, + { + name: "list_loons", + pack: null, + description: + "List the stored loon macro library: names, versions, parameter docs with units and " + + "example values. Use before call_loon or create_cad_loon with use_loons.", + inputSchema: { type: "object" as const, properties: {} }, + handler: (_args, ctx) => listLoonsTool(ctx.user), + behavior: behavior({}), + }, +]; diff --git a/packages/mcp/src/tools/loon.ts b/packages/mcp/src/tools/loon.ts index e4c03afe3..0a4ad527b 100644 --- a/packages/mcp/src/tools/loon.ts +++ b/packages/mcp/src/tools/loon.ts @@ -5,6 +5,7 @@ import type { Engine } from "@vcad/engine"; import { toVCode } from "@vcad/ir"; import { appendIntegrity, computeIntegrity } from "./integrity.js"; +import { hydrateMacros, macroPrelude, type InlineLoon } from "./loon-macros.js"; import { behavior, type ToolDef } from "./tool-def.js"; import type { ToolResult } from "./tool-result.js"; @@ -16,6 +17,32 @@ export const createCadLoonSchema = { type: "string" as const, description: "Loon source code defining CAD geometry", }, + use_loons: { + type: "array" as const, + items: { type: "string" as const }, + description: + "Stored macro names (see list_loons) to prepend as a library — " + + "their [let [fn ...]] definitions become callable from " + + "`source`, exactly like the stdlib. List dependencies before " + + "dependents.", + }, + loons: { + type: "array" as const, + description: + "STATELESS macro library: macros passed by value (the `macro` " + + "records define_loon returns: {name, source}). Prepended like " + + "use_loons but with no server-side registry dependency — immune " + + "to serverless cold starts. Names here also satisfy use_loons.", + items: { + type: "object" as const, + required: ["name", "source"], + properties: { + name: { type: "string" as const }, + source: { type: "string" as const }, + params: { type: "array" as const }, + }, + }, + }, format: { type: "string" as const, enum: ["vcode", "json"], @@ -27,15 +54,31 @@ export const createCadLoonSchema = { interface CreateLoonInput { source: string; + use_loons?: string[]; + loons?: InlineLoon[]; format?: "vcode" | "json"; } +/** Compose the effective program: macro prelude (inline `loons` win over + * the registry) + user source. Inline macros not named in use_loons are + * prepended too — passing `loons` alone is sufficient. */ +export function composeLoonProgram(input: unknown): string { + const { source, use_loons, loons } = input as CreateLoonInput; + const names = [ + ...(use_loons ?? []), + ...(loons ?? []).map((m) => m.name).filter((n) => !use_loons?.includes(n)), + ]; + if (!names.length) return source; + return `${macroPrelude(names, loons)}\n\n${source}`; +} + /** Evaluate loon source and return a CAD document. */ export function createCadLoon( input: unknown, engine: Engine, ): { content: Array<{ type: "text"; text: string }> } { - const { source, format = "vcode" } = input as CreateLoonInput; + const { format = "vcode" } = input as CreateLoonInput; + const source = composeLoonProgram(input); const doc = engine.evalVcadSource(source); if (!doc) { @@ -69,13 +112,21 @@ export const toolDefs: ToolDef[] = [ "Let bindings: [let body [cube 50 30 5]]\n" + "Scene: [root solid \"material-name\"]", inputSchema: createCadLoonSchema, - handler: (args, ctx) => { + handler: async (args, ctx) => { + // Hydrate any by-name macros from the durable per-user store before + // composing (cold serverless instances start with an empty registry). + const useLoons = Array.isArray(args.use_loons) + ? (args.use_loons as string[]) + : undefined; + if (useLoons?.length) { + await hydrateMacros(ctx.user, useLoons).catch(() => {}); + } const result = createCadLoon(args, ctx.engine) as ToolResult; // Attach the integrity certificate to the largest mutation of all: // authoring a whole document. The loon evaluation is cheap relative to // the mesh evaluation computeIntegrity runs anyway. try { - const doc = ctx.engine.evalVcadSource(String(args.source ?? "")); + const doc = ctx.engine.evalVcadSource(composeLoonProgram(args)); if (doc) { const integrity = computeIntegrity(doc, ctx.engine); if (integrity) appendIntegrity(result, integrity); diff --git a/packages/mcp/src/tools/physics.ts b/packages/mcp/src/tools/physics.ts new file mode 100644 index 000000000..fbee21e15 --- /dev/null +++ b/packages/mcp/src/tools/physics.ts @@ -0,0 +1,333 @@ +/** + * predict_physics tool — two-tier static structural analysis. + * + * The fast inner loop of physics-validated generation: voxel FEA over a + * part's volume (or a box) under given loads and supports. `fidelity` + * picks the tier — `predict` answers in ~100 ms at coarse resolution and + * stamps every claim `basis: predicted`; `verify` re-runs the SAME solver + * at fine resolution and stamps `basis: verified`. A receipt whose passing + * claims rest on predictions rolls up `provisional`, never `pass` + * (crates/vcad-receipt) — predictions steer, verification certifies. + */ + +import type { Engine, StaticAnalysisSpec, StaticAnalysisResult } from "@vcad/engine"; +import type { DesignReceipt, ReceiptClaim, OracleRef } from "@vcad/ir"; +import { getSession } from "./session.js"; +import { behavior, type ToolDef } from "./tool-def.js"; +import { resolvePartMesh } from "./topopt.js"; +import { RECEIPT_SCHEMA, summarize, unverifiableClaim } from "../receipt-unified.js"; + +const PHYSICS_DOMAIN = "mechanical"; +const ORACLE: OracleRef = { id: "vcad-kernel-topopt/static-fea", version: "0.9.4" }; + +/** Resolution per fidelity tier. Same solver; the grid is the only dial. */ +const TIER_RESOLUTION = { predict: 32, verify: 72 } as const; + +const regionSchema = { + type: "object" as const, + required: ["min", "max"], + properties: { + min: { + type: "array" as const, + items: { type: "number" as const }, + minItems: 3, + maxItems: 3, + description: "Minimum corner [x, y, z] in mm.", + }, + max: { + type: "array" as const, + items: { type: "number" as const }, + minItems: 3, + maxItems: 3, + description: "Maximum corner [x, y, z] in mm.", + }, + }, +}; + +export const predictPhysicsSchema = { + type: "object" as const, + required: ["loads", "supports"], + properties: { + document_id: { + type: "string" as const, + description: "Session document. Required with `part`.", + }, + part: { + type: "string" as const, + description: + "Part id or name to analyze (its evaluated volume is voxelized). " + + "Mutually exclusive with `domain_box`.", + }, + domain_box: { + ...regionSchema, + description: + "Analyze a solid axis-aligned box (mm, world frame, Z-up) instead " + + "of a part. Mutually exclusive with `part`.", + }, + loads: { + type: "array" as const, + minItems: 1, + description: + "Loads: total force vectors (N) distributed over the grid nodes in " + + "each world-frame box region. A zero-thickness box selects the " + + "nearest plane of nodes.", + items: { + type: "object" as const, + required: ["region", "force"], + properties: { + region: regionSchema, + force: { + type: "array" as const, + items: { type: "number" as const }, + minItems: 3, + maxItems: 3, + description: "Total force [fx, fy, fz] in N.", + }, + }, + }, + }, + supports: { + type: "array" as const, + minItems: 1, + description: "Fixed (anchored) regions.", + items: { + type: "object" as const, + required: ["region"], + properties: { + region: regionSchema, + fix: { + type: "array" as const, + items: { type: "boolean" as const }, + minItems: 3, + maxItems: 3, + description: + "Which translations are fixed [x, y, z]; default all true.", + }, + }, + }, + }, + fidelity: { + type: "string" as const, + enum: ["predict", "verify"], + description: + "`predict` (default): coarse fast solve, claims stamped " + + "basis=predicted — good enough to steer a design. `verify`: fine " + + "solve with the same oracle, claims stamped basis=verified — good " + + "enough to certify. A receipt passing only on predicted claims " + + "reads `provisional`, never `pass`.", + }, + resolution: { + type: "number" as const, + description: + "Override voxels along the longest axis (predict=32, verify=72 by " + + "default). Below ~4 elements through the thinnest section, bending " + + "results are unreliable.", + }, + youngs_modulus_mpa: { + type: "number" as const, + description: "Young's modulus in MPa. Default 69000 (6061 aluminum).", + }, + poisson: { + type: "number" as const, + description: "Poisson's ratio. Default 0.33.", + }, + max_displacement_mm: { + type: "number" as const, + description: + "Optional limit: assert max displacement ≤ this (claim " + + "physics.static.displacement).", + }, + max_von_mises_mpa: { + type: "number" as const, + description: + "Optional limit: assert max von Mises stress ≤ this (claim " + + "physics.static.stress). E.g. yield/safety-factor.", + }, + }, +}; + +interface PhysicsArgs { + document_id?: string; + part?: string; + domain_box?: { min: [number, number, number]; max: [number, number, number] }; + loads?: StaticAnalysisSpec["loads"]; + supports?: StaticAnalysisSpec["supports"]; + fidelity?: "predict" | "verify"; + resolution?: number; + youngs_modulus_mpa?: number; + poisson?: number; + max_displacement_mm?: number; + max_von_mises_mpa?: number; +} + +const round5 = (v: number) => Number(v.toPrecision(5)); + +function limitClaim( + id: string, + description: string, + subject: string | undefined, + limit: number, + actual: number, + unit: string, + basis: "predicted" | "verified", + converged: boolean, +): ReceiptClaim { + if (!converged || !Number.isFinite(actual)) { + return { + ...unverifiableClaim(id, PHYSICS_DOMAIN, description, ORACLE, "FE solve did not converge"), + basis, + subject, + }; + } + return { + id, + domain: PHYSICS_DOMAIN, + description, + subject, + oracle: ORACLE, + verdict: actual <= limit ? "pass" : "fail", + basis, + predicted: { value: limit, unit }, + measured: { value: round5(actual), unit }, + }; +} + +export function predictPhysicsTool( + args: Record, + engine: Engine, +): { content: Array<{ type: "text"; text: string }> } { + const a = args as PhysicsArgs; + + if (!a.loads?.length) throw new Error("predict_physics: `loads` required"); + if (!a.supports?.length) throw new Error("predict_physics: `supports` required"); + if (!!a.part === !!a.domain_box) { + throw new Error("predict_physics: pass exactly one of `part` or `domain_box`"); + } + if (a.part && !a.document_id) { + throw new Error("predict_physics: `part` requires `document_id`"); + } + const fidelity = a.fidelity ?? "predict"; + const basis = fidelity === "verify" ? "verified" : "predicted"; + + const spec: StaticAnalysisSpec = { + loads: a.loads, + supports: a.supports, + resolution: a.resolution ?? TIER_RESOLUTION[fidelity], + youngs_modulus_mpa: a.youngs_modulus_mpa, + poisson: a.poisson, + }; + + let documentId: string | undefined; + let subject: string | undefined; + const started = performance.now(); + let result: StaticAnalysisResult; + if (a.part) { + documentId = String(a.document_id); + const doc = getSession(documentId); + const resolved = resolvePartMesh(doc, engine, a.part); + subject = `part:${resolved.name ?? a.part}`; + result = engine.analyzeStaticsMesh(resolved.mesh, spec); + } else { + // Box runs are pure computation — no session is touched or minted. + if (a.document_id) documentId = String(a.document_id); + const box = a.domain_box!; + result = engine.analyzeStaticsBox(box.min, box.max, spec); + subject = "domain_box"; + } + const solveMs = Math.round(performance.now() - started); + + const claims: ReceiptClaim[] = []; + if (a.max_displacement_mm !== undefined) { + claims.push( + limitClaim( + "physics.static.displacement", + `max displacement under load ≤ ${a.max_displacement_mm} mm`, + subject, + a.max_displacement_mm, + result.maxDisplacementMm, + "mm", + basis, + result.converged, + ), + ); + } + if (a.max_von_mises_mpa !== undefined) { + claims.push( + limitClaim( + "physics.static.stress", + `max von Mises stress ≤ ${a.max_von_mises_mpa} MPa`, + subject, + a.max_von_mises_mpa, + result.maxVonMisesMpa, + "MPa", + basis, + result.converged, + ), + ); + } + + const receipt: DesignReceipt | undefined = claims.length + ? { schema: RECEIPT_SCHEMA, document_id: documentId, claims } + : undefined; + + return { + content: [ + { + type: "text", + text: JSON.stringify( + { + document_id: documentId, + fidelity, + basis, + solve_ms: solveMs, + analysis: { + max_displacement_mm: round5(result.maxDisplacementMm), + max_displacement_at: result.maxDisplacementAt.map(round5), + max_von_mises_mpa: round5(result.maxVonMisesMpa), + max_stress_at: result.maxStressAt.map(round5), + compliance_n_mm: round5(result.compliance), + grid: result.grid, + voxel_size_mm: round5(result.voxelSizeMm), + converged: result.converged, + }, + ...(receipt + ? { receipt, summary: summarize(receipt) } + : { + note: + "No limits asserted — pass max_displacement_mm and/or " + + "max_von_mises_mpa to get receipt claims.", + }), + ...(fidelity === "predict" + ? { + next: + "Estimates only (basis=predicted → summary.verdict=" + + "provisional). Re-run with fidelity=\"verify\" to certify.", + } + : {}), + }, + null, + 2, + ), + }, + ], + }; +} + +export const toolDefs: ToolDef[] = [ + { + name: "predict_physics", + pack: null, + description: + "Fast static structural analysis (voxel FEA): max displacement, max von Mises stress, and " + + "compliance for a part or box under world-frame loads (N) and supports, in ~100 ms at " + + "fidelity=predict. Pass max_displacement_mm / max_von_mises_mpa limits to get receipt " + + "claims: predict-tier claims carry basis=predicted and roll up as a PROVISIONAL receipt — " + + "use them to iterate on a design cheaply. When the design settles, re-run with " + + "fidelity=verify (same solver, fine grid) to upgrade the claims to basis=verified and a " + + "certifiable pass. Loads/supports are box regions (mm, Z-up); zero-thickness boxes select " + + "a face. Voxel FEA smears stress concentrations — treat stress as an estimate near fillets.", + inputSchema: predictPhysicsSchema, + handler: (args, ctx) => predictPhysicsTool(args, ctx.engine), + behavior: behavior({}), + }, +]; diff --git a/packages/mcp/src/tools/tool-metadata.ts b/packages/mcp/src/tools/tool-metadata.ts index 16508f351..8d69c2be6 100644 --- a/packages/mcp/src/tools/tool-metadata.ts +++ b/packages/mcp/src/tools/tool-metadata.ts @@ -240,6 +240,10 @@ export const TOOL_METADATA: Record = { }), }, topology_optimize: { title: "Topology Optimize", annotations: RW }, + predict_physics: { title: "Predict Physics", annotations: RO }, + define_loon: { title: "Define Loon Macro", annotations: RW }, + call_loon: { title: "Call Loon Macro", annotations: RW }, + list_loons: { title: "List Loon Macros", annotations: RO }, list_footprints: { title: "List Footprints", annotations: RO }, search_footprints: { title: "Search Footprints", annotations: RO }, get_pad_positions: { title: "Get Pad Positions", annotations: RO }, diff --git a/packages/mcp/src/tools/topopt.ts b/packages/mcp/src/tools/topopt.ts index 37e41ecf3..5987585f5 100644 --- a/packages/mcp/src/tools/topopt.ts +++ b/packages/mcp/src/tools/topopt.ts @@ -164,7 +164,7 @@ interface TopoArgs { } /** Resolve a part (by root id or name) to its evaluated, placed mesh. */ -function resolvePartMesh( +export function resolvePartMesh( doc: Document, engine: Engine, wanted: string, diff --git a/packages/mcp/src/trust-boundary.ts b/packages/mcp/src/trust-boundary.ts new file mode 100644 index 000000000..990bdae74 --- /dev/null +++ b/packages/mcp/src/trust-boundary.ts @@ -0,0 +1,200 @@ +/** + * Trust boundary for the commerce plane (docs/trust-boundary.md). + * + * vcad agents ingest untrusted content (imported STEP/KiCad/Eagle files, + * part descriptions, datasheets) and also hold spend authority + * (authorize_spend / place_order). This module is the mechanical guard + * between the two: a synchronous, pure pre-dispatch check that runs before + * any commerce tool handler and rejects argument shapes an injection would + * need — regardless of what the model was convinced to do. + * + * The rules are deliberately dumb and enforceable, not heuristic: + * 1. Money-plane tools accept opaque ids only. Every id-shaped argument + * must match a safe charset; a "part number" or order id carrying a URL, + * whitespace, or control characters is refused outright. + * 2. Artifact handles must point at OUR artifact store. A bare `art_…` id + * or a relative `/artifacts/…` path is fine; a full URL is only accepted + * on an allowlisted vcad host. A poisoned document that plants + * `https://evil.example/artifacts/art_x` never reaches resolution. + * 3. Free-text that travels to the fab (ship_to, material, finish) must be + * plain: no URLs, no control characters, bounded length. An address is + * not a place for instructions. + * + * Fail-closed: the guard refuses on violation with a stable, greppable + * `TRUST_BOUNDARY:` message; it never rewrites arguments. + */ + +/** Tools the guard applies to (the commerce plane). */ +export const COMMERCE_TOOLS: ReadonlySet = new Set([ + "quote_manufacturing", + "authorize_spend", + "place_order", +]); + +/** Arguments that must be opaque ids, per tool. */ +const ID_FIELDS: Record = { + quote_manufacturing: ["document_id"], + authorize_spend: ["order_id"], + place_order: ["order_id", "authorization_id", "idempotency_key"], +}; + +/** Free-text fields that travel to the fab, per tool. */ +const FAB_TEXT_FIELDS: Record = { + quote_manufacturing: ["material", "finish"], +}; + +/** Artifact-handle fields, per tool. */ +const ARTIFACT_FIELDS: Record = { + quote_manufacturing: ["fab_artifact_id"], + place_order: ["fab_artifact_id"], +}; + +/** Opaque-id charset: what our own tools mint (uuid/art_/ord_/auth_ …). */ +const SAFE_ID = /^[A-Za-z0-9._:-]{1,128}$/; + +/** Hosts an artifact_url may name. Everything else is refused. */ +const ARTIFACT_HOSTS: ReadonlySet = new Set([ + "mcp.vcad.io", + "vcad.io", + "www.vcad.io", + "localhost", + "127.0.0.1", +]); + +const SHIP_TO_MAX_FIELD_LEN = 200; +const SHIP_TO_MAX_FIELDS = 24; + +/** Result of a boundary check. */ +export interface BoundaryVerdict { + ok: boolean; + /** Present when `ok` is false; starts with `TRUST_BOUNDARY:`. */ + reason?: string; +} + +const pass: BoundaryVerdict = { ok: true }; + +function refuse(reason: string): BoundaryVerdict { + return { ok: false, reason: `TRUST_BOUNDARY: ${reason}` }; +} + +/* eslint-disable no-control-regex */ +const CONTROL_CHARS = /[\x00-\x1f\x7f]/; +/* eslint-enable no-control-regex */ +const URL_SCHEME = /[a-z][a-z0-9+.-]*:\/\//i; + +function hasUrl(s: string): boolean { + return URL_SCHEME.test(s) || /\bwww\.[a-z0-9-]+\.[a-z]{2,}/i.test(s); +} + +/** + * Is this handle allowed to reach artifact resolution? Bare `art_…` ids and + * relative `/artifacts/…` paths always are; absolute URLs only on our hosts. + */ +export function isAllowedArtifactHandle(handle: string): boolean { + if (CONTROL_CHARS.test(handle)) return false; + if (SAFE_ID.test(handle) && !/^\.+$/.test(handle)) return true; // bare id + if (handle.startsWith("/")) { + // Relative path: /artifacts/[/] only. Dot-only segments would + // be path traversal, and the id charset admits them — refuse explicitly. + if (!/^\/artifacts\/[A-Za-z0-9._:-]+(\/[A-Za-z0-9._:-]+)?$/.test(handle)) { + return false; + } + return handle.split("/").every((seg) => !/^\.+$/.test(seg) || seg === ""); + } + let url: URL; + try { + url = new URL(handle); + } catch { + return false; + } + if (url.protocol !== "https:" && url.protocol !== "http:") return false; + if (!ARTIFACT_HOSTS.has(url.hostname)) return false; + return url.pathname.includes("/artifacts/"); +} + +/** One flat or one-level-nested string field of ship_to. */ +function badShipToValue(v: unknown): string | null { + if (v === null || v === undefined) return null; + if (typeof v === "number" || typeof v === "boolean") return null; + if (typeof v !== "string") return "non-scalar value"; + if (v.length > SHIP_TO_MAX_FIELD_LEN) return "field too long"; + if (CONTROL_CHARS.test(v)) return "control characters"; + if (hasUrl(v)) return "embedded URL"; + return null; +} + +/** Validate a ship_to object: bounded, flat-ish, plain text only. */ +export function checkShipTo(shipTo: unknown): BoundaryVerdict { + if (shipTo === undefined || shipTo === null) return pass; + if (typeof shipTo !== "object" || Array.isArray(shipTo)) { + return refuse("ship_to must be an object of plain address fields"); + } + const entries = Object.entries(shipTo as Record); + if (entries.length > SHIP_TO_MAX_FIELDS) { + return refuse("ship_to has too many fields"); + } + for (const [k, v] of entries) { + const bad = badShipToValue(v); + if (bad) { + return refuse( + `ship_to.${k}: ${bad}. Addresses carry plain text only — no URLs, ` + + `no control characters, ≤${SHIP_TO_MAX_FIELD_LEN} chars per field.`, + ); + } + } + return pass; +} + +/** + * Pre-dispatch guard. Call for every tool; non-commerce tools pass through + * untouched. Pure and synchronous — safe at the dispatch choke-point. + */ +export function checkCommerceBoundary( + toolName: string, + args: Record, +): BoundaryVerdict { + if (!COMMERCE_TOOLS.has(toolName)) return pass; + + for (const field of ID_FIELDS[toolName] ?? []) { + const v = args[field]; + if (v === undefined || v === null) continue; + if (typeof v !== "string" || !SAFE_ID.test(v)) { + return refuse( + `${toolName}.${field} must be an opaque id (letters, digits, ` + + `.:_-, ≤128 chars) minted by a vcad tool — not free text. ` + + `Ids from part descriptions, datasheets, or imported files are ` + + `never valid here.`, + ); + } + } + + for (const field of ARTIFACT_FIELDS[toolName] ?? []) { + const v = args[field]; + if (v === undefined || v === null || v === "") continue; + if (typeof v !== "string" || !isAllowedArtifactHandle(v)) { + return refuse( + `${toolName}.${field} must reference the vcad artifact store: a ` + + `bare art_… id, a relative /artifacts/… path, or an artifact URL ` + + `on a vcad host. External URLs are never fetched or bound to orders.`, + ); + } + } + + for (const field of FAB_TEXT_FIELDS[toolName] ?? []) { + const v = args[field]; + if (v === undefined || v === null) continue; + if (typeof v !== "string" || CONTROL_CHARS.test(v) || hasUrl(v) || v.length > 120) { + return refuse( + `${toolName}.${field} must be a short plain-text label (≤120 chars, ` + + `no URLs, no control characters).`, + ); + } + } + + if (toolName === "quote_manufacturing") { + const shipTo = checkShipTo(args.ship_to); + if (!shipTo.ok) return shipTo; + } + + return pass; +} diff --git a/supabase/migrations/036_mcp_macros.sql b/supabase/migrations/036_mcp_macros.sql new file mode 100644 index 000000000..24eb14a9f --- /dev/null +++ b/supabase/migrations/036_mcp_macros.sql @@ -0,0 +1,46 @@ +-- Per-user durable storage for the agent loon-macro library. +-- +-- WHY: macros defined via define_loon live in a process-warm Map plus local +-- JSON files (packages/mcp/src/tools/loon-macros.ts) — on the serverless +-- deploy a macro defined on one instance is invisible everywhere else, and +-- dies with the instance. Same failure class — and same fix — as sessions +-- (SupabaseSessionStore) and artifacts (migration 033): a durable table the +-- MCP server hydrates on miss. Unlike artifacts, macros are user-scoped and +-- permanent (a library, not a cache): keyed (user_id, name), no TTL. +-- +-- The MCP server writes with the service role (bypasses RLS) always scoping +-- user_id to the verified caller; RLS mirrors `documents` so a future +-- signed-in web UI can read/manage the user's own library directly. + +create table if not exists mcp_macros ( + user_id uuid not null references auth.users (id) on delete cascade, + -- kebab-case macro name; also the loon function the source defines. + name text not null, + -- Monotone version, bumped by the server on redefinition. + version integer not null default 1, + description text not null default '', + -- [{name, description?, example, unit?}] — ordered parameter docs. + params jsonb not null default '[]'::jsonb, + -- The loon source: [let [fn [params...] ...]] (+ helpers). + source text not null, + -- Reserved for the certify_loon rung: a DesignReceipt (vcad.receipt/1) + -- whose claims cover the macro's parameter range at verify tier. Null = + -- uncertified (smoke-tested only). + receipt jsonb, + updated_at timestamptz not null default now(), + primary key (user_id, name) +); + +alter table mcp_macros enable row level security; + +create policy "Users can view their own macros" on mcp_macros + for select using (auth.uid() = user_id); + +create policy "Users can insert their own macros" on mcp_macros + for insert with check (auth.uid() = user_id); + +create policy "Users can update their own macros" on mcp_macros + for update using (auth.uid() = user_id); + +create policy "Users can delete their own macros" on mcp_macros + for delete using (auth.uid() = user_id);