From f499734a55149c49eb9a9da4f51772de994dffa4 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 5 Oct 2026 06:28:27 +0000 Subject: [PATCH 1/5] USD import and texture caches: pay off duplicated logic - attrs.rs: one value_at/prim_value plus decode_* helpers replace ~45 hand-written get_at:: matches. Every f32 reader now accepts Half, the OpenPBR shader inputs accept Vec3d and int-authored bools. - prune_reason(): the abstract/inactive/purpose/invisible rule written once, shared by the traversal, the placement count and the prototype walk. - compose_with_parent(): the reset-xform-stack rule at one site. - assets.rs: one asset-path rule (authoring-layer anchoring everywhere, lights included) and one memoized, timed cached_asset() for every host decode. - camera.rs: CameraFrame, the one read build_camera and screen_projection both derive from. - existing_tiles(): the one 10x10 UDIM sweep (preload, streaming, .tx conversion, maketx); StreamingTexture::open loses its expand callback. - texture_cache.rs: the FIFO microcache set and MiB conversion shared by the .tx tile cache and Ptex streaming; drop the unused legacy env wrappers (ptex_*_from_env, ptex_stream_enabled, budget_from_env). All sample scenes bit-identical at 16 spp. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_019nwge6NCTPhuk1VPuZucRF --- crates/crust-assets/src/lib.rs | 65 ++--- crates/crust-assets/src/ptex_stream.rs | 73 ++--- crates/crust-assets/src/texture_cache.rs | 78 ++++++ crates/crust-assets/src/tiled/cache.rs | 42 +-- crates/crust-assets/src/tiled/stream.rs | 123 +++------ crates/crust-assets/src/uv_texture/mod.rs | 40 +-- crates/crust-assets/src/uv_texture/udim.rs | 39 ++- .../crust-core/src/scene/usd_import/assets.rs | 169 ++++++++++++ .../crust-core/src/scene/usd_import/attrs.rs | 216 ++++++++------- .../crust-core/src/scene/usd_import/camera.rs | 122 +++++---- .../src/scene/usd_import/instancing.rs | 92 ++----- .../src/scene/usd_import/light_links.rs | 33 +-- .../crust-core/src/scene/usd_import/lights.rs | 159 +++++------ .../src/scene/usd_import/materials.rs | 252 ++---------------- .../crust-core/src/scene/usd_import/mesh.rs | 72 ++--- crates/crust-core/src/scene/usd_import/mod.rs | 175 ++++++------ .../src/scene/usd_import/preview.rs | 60 +---- .../src/scene/usd_import/products.rs | 28 +- .../src/scene/usd_import/settings.rs | 7 +- .../crust-core/src/scene/usd_import/shapes.rs | 51 +--- .../crust-core/src/scene/usd_import/xform.rs | 67 ++--- crates/crust-render/examples/maketx.rs | 12 +- 22 files changed, 872 insertions(+), 1103 deletions(-) create mode 100644 crates/crust-assets/src/texture_cache.rs create mode 100644 crates/crust-core/src/scene/usd_import/assets.rs diff --git a/crates/crust-assets/src/lib.rs b/crates/crust-assets/src/lib.rs index 3d0d3d20..663d3fd4 100644 --- a/crates/crust-assets/src/lib.rs +++ b/crates/crust-assets/src/lib.rs @@ -31,6 +31,7 @@ mod image_file; mod mip_filter; mod ptex_stream; mod ptex_texture; +mod texture_cache; pub mod tiled; mod uv_texture; @@ -40,18 +41,32 @@ pub use ies::{load_ies, parse_ies}; pub use ptex_stream::{ DEFAULT_CACHE_MB as PTEX_DEFAULT_CACHE_MB, DEFAULT_STREAM_MIN_MB as PTEX_DEFAULT_STREAM_MIN_MB, MICRO_SLOTS as PTEX_MICRO_SLOTS, MipSpace as PtexMipSpace, PtexStream, - StreamStats as PtexStreamStats, cache_budget_from_env as ptex_cache_budget_from_env, - micro_reserve as ptex_micro_reserve, micro_retained_bytes as ptex_micro_retained_bytes, - micro_slot_max as ptex_micro_slot_max, micro_thread_bytes as ptex_micro_thread_bytes, - micro_threads as ptex_micro_threads, mip_space_from_env as ptex_mip_space_from_env, - stream_enabled as ptex_stream_enabled, - stream_min_bytes_from_env as ptex_stream_min_bytes_from_env, + StreamStats as PtexStreamStats, micro_reserve as ptex_micro_reserve, + micro_retained_bytes as ptex_micro_retained_bytes, micro_slot_max as ptex_micro_slot_max, + micro_thread_bytes as ptex_micro_thread_bytes, micro_threads as ptex_micro_threads, }; pub use ptex_texture::{ DEFAULT_MAX_LOG2, PtexColor, max_log2_from_env, max_log2_from_env_opt, read_channel, }; pub use uv_texture::{DEFAULT_MAX_EDGE, UvTexture}; +/// The files a texture path names: every `` / `` tile on +/// disk (the 10x10 UDIM sweep every texture reader shares), or the one image +/// when it exists. +pub fn texture_files(path: &Path) -> Vec { + let name = path.to_string_lossy(); + if name.contains("") || name.contains("") { + uv_texture::existing_tiles(&name) + .into_iter() + .map(|t| t.path) + .collect() + } else if path.exists() { + vec![path.to_path_buf()] + } else { + Vec::new() + } +} + use crust_core::{ AssetLoader, ColorSpace, EnvironmentMap, IesProfile, LightTexture, PtexTexture, ResolvedColorSpace, Texture2D, @@ -282,7 +297,7 @@ impl FileAssets { debug!( "Textures stream from a .tx beside them when one exists, through a {:.0} MiB \ tile cache (CRUST_TEX_STREAM=0 preloads everything)", - budget as f64 / (1024.0 * 1024.0) + texture_cache::bytes_to_mib(budget) ); } let ptex_streaming = config.ptex_stream; @@ -290,7 +305,7 @@ impl FileAssets { if ptex_streaming { info!( "Streaming Ptex with a {:.0} MiB cache", - ptex_stream::budget_bytes(&config) as f64 / (1024.0 * 1024.0) + texture_cache::bytes_to_mib(ptex_stream::budget_bytes(&config) as u64) ); // Said at construction rather than per texture, because under // the default policy it is the line that explains a render where @@ -370,30 +385,6 @@ impl FileAssets { ) } - /// The files a texture path names: every `` / `` tile on - /// disk, or the one image. - fn tile_sources(path: &Path) -> Vec { - let name = path.to_string_lossy(); - if name.contains("") || name.contains("") { - let mut tiles = Vec::new(); - for v in 0..10u32 { - for u in 0..10u32 { - if let Some(p) = - uv_texture::expand_token(&name, u, v).map(std::path::PathBuf::from) - && p.exists() - { - tiles.push(p); - } - } - } - tiles - } else if path.exists() { - vec![path.to_path_buf()] - } else { - Vec::new() - } - } - /// Whether a complete set of `.tx` siblings stands beside `path`'s tiles, /// converting the missing and stale ones first under `--auto-tx`. /// @@ -412,7 +403,7 @@ impl FileAssets { if is_tx { return true; } - let sources = Self::tile_sources(path); + let sources = texture_files(path); if sources.is_empty() { return false; } @@ -687,11 +678,7 @@ impl FileAssets { .filter(|c| (c.as_path() == path) == (which == Candidates::Source)); for candidate in candidates { let started = Instant::now(); - let Some(tex) = - tiled::StreamingTexture::open(&candidate, space, self.cache.clone(), |u, v| { - let name = candidate.to_string_lossy(); - uv_texture::expand_token(&name, u, v).map(std::path::PathBuf::from) - }) + let Some(tex) = tiled::StreamingTexture::open(&candidate, space, self.cache.clone()) else { continue; }; @@ -906,7 +893,7 @@ impl AssetLoader for FileAssets { // `DEFAULT_STREAM_MIN_MB` for the island distribution that // makes this necessary rather than tidy. let would = tex.preload_bytes(self.preload_max_log2()); - let floor = self.config.ptex_stream_min_mb * 1024 * 1024; + let floor = ptex_stream::stream_min_bytes(&self.config); if would < floor { why = PreloadReason::TooSmall; debug!( diff --git a/crates/crust-assets/src/ptex_stream.rs b/crates/crust-assets/src/ptex_stream.rs index 19ac7218..7aec0dca 100644 --- a/crates/crust-assets/src/ptex_stream.rs +++ b/crates/crust-assets/src/ptex_stream.rs @@ -38,6 +38,8 @@ use crate::ptex_texture::ptex_space; use crate::read_channel; use crust_core::{ColorSpace, PtexTexture, Vec3A}; use std::path::Path; + +use crate::texture_cache::{Ways, mib_to_bytes}; use std::sync::atomic::{AtomicU32, Ordering}; /// Default cache budget, in MiB. @@ -49,20 +51,10 @@ use std::sync::atomic::{AtomicU32, Ordering}; /// working set is the frame rather than a locality window. pub const DEFAULT_CACHE_MB: usize = crust_core::config::DEFAULT_CACHE_MB; -/// `CRUST_PTEX_CACHE_MB` as parsed into [`crust_core::config()`], as a byte -/// count. -pub fn cache_budget_from_env() -> usize { - budget_bytes(crust_core::config()) -} - -/// [`cache_budget_from_env`] for a given configuration. +/// The render's whole Ptex budget `config` asks for (`CRUST_PTEX_CACHE_MB`), +/// in bytes. pub fn budget_bytes(config: &crust_core::Config) -> usize { - config.ptex_cache_mb.get() * 1024 * 1024 -} - -/// Is the streaming backend on? `CRUST_PTEX_STREAM=1` turns it on. -pub fn stream_enabled() -> bool { - crust_core::config().ptex_stream + mib_to_bytes(config.ptex_cache_mb.get() as u64) as usize } /// Default admission threshold: a texture streams only if **preloading** it @@ -96,10 +88,10 @@ pub fn stream_enabled() -> bool { /// what reproduces the even-split behaviour for comparison. pub const DEFAULT_STREAM_MIN_MB: usize = crust_core::config::DEFAULT_PTEX_STREAM_MIN_MB; -/// `CRUST_PTEX_STREAM_MIN_MB` as parsed into [`crust_core::config()`], as a -/// byte count. See [`DEFAULT_STREAM_MIN_MB`]. -pub fn stream_min_bytes_from_env() -> usize { - crust_core::config().ptex_stream_min_mb * 1024 * 1024 +/// The admission threshold `config` asks for (`CRUST_PTEX_STREAM_MIN_MB`), in +/// bytes. See [`DEFAULT_STREAM_MIN_MB`]. +pub(crate) fn stream_min_bytes(config: &crust_core::Config) -> usize { + mib_to_bytes(config.ptex_stream_min_mb as u64) as usize } /// Which mip chain a streamed texture is allowed to read — the reasoning is @@ -107,12 +99,6 @@ pub fn stream_min_bytes_from_env() -> usize { /// into. pub use crust_core::PtexMipSpace as MipSpace; -/// `CRUST_PTEX_STREAM_MIPSPACE` as parsed into [`crust_core::config()`]. See -/// [`MipSpace`]. -pub fn mip_space_from_env() -> MipSpace { - crust_core::config().ptex_mip_space -} - /// Mip levels a face of resolution `res` holds, halving each axis to a floor /// of one texel. /// @@ -190,7 +176,7 @@ struct TileId { /// and leave the interior at 0.999, for one extra `Option` pair per thread /// and a linear scan that finds its hit at index 0 either way. Pinned by /// `the_microcache_absorbs_most_taps`. -type MicroSlots = [Option<(TileId, ptex::PixelData)>; MICRO_SLOTS]; +type MicroSlots = Ways; /// See [`MicroSlots`]: four is the corner case's tap count, not a round /// number. Dropping it to two is what the 0.000 above measures. @@ -277,18 +263,12 @@ pub fn micro_retained_bytes() -> u64 { /// same time — another test in the same binary, for one — moves it too. A /// check on what one sequence of lookups retained has to read this instead. pub fn micro_thread_bytes() -> u64 { - MICRO.with(|m| { - m.borrow() - .iter() - .flatten() - .map(|(_, data)| data.len() as u64) - .sum() - }) + MICRO.with(|m| m.borrow().values().map(|data| data.len() as u64).sum()) } thread_local! { static MICRO: std::cell::RefCell = - const { std::cell::RefCell::new([const { None }; MICRO_SLOTS]) }; + const { std::cell::RefCell::new(MicroSlots::EMPTY) }; } /// Distinguishes textures in [`TileId`]. Wraps only after 4 billion `.ptx` @@ -630,10 +610,7 @@ impl PtexStream { let mut f = Some(f); let hit = MICRO.with(|m| { let slots = m.borrow(); - let idx = slots - .iter() - .position(|s| matches!(s, Some((k, _)) if *k == id))?; - let (_, data) = slots[idx].as_ref()?; + let data = slots.get(&id)?; Some(f.take()?(data)) }); if let Some(r) = hit { @@ -658,18 +635,11 @@ impl PtexStream { if data.len() > self.micro_max { return Some(r); } - MICRO.with(|m| { - let mut slots = m.borrow_mut(); - // Take the entry about to fall off the end *before* rotating, so - // its bytes leave the accounting with it; `rotate_right` then - // puts that hole in front for the new tile. - if let Some((_, old)) = slots[MICRO_SLOTS - 1].take() { - MICRO_BYTES.sub(old.len() as u64); - } - slots.rotate_right(1); - MICRO_BYTES.add(data.len() as u64); - slots[0] = Some((id, data)); - }); + // The entry that falls off the end leaves the accounting with it. + MICRO_BYTES.add(data.len() as u64); + if let Some((_, old)) = MICRO.with(|m| m.borrow_mut().push(id, data)) { + MICRO_BYTES.sub(old.len() as u64); + } Some(r) } @@ -882,10 +852,9 @@ mod tests { } #[test] - fn a_budget_is_read_from_the_environment_and_a_bad_one_falls_back() { - // No env mutation: `cache_budget_from_env` is the wrapper, and the - // policy it wraps is what matters. Kept as a compile-time check that - // the default is stated in one place and in MiB. + fn the_default_budget_is_stated_once_and_in_mib() { + // A compile-time check that the default is stated in one place and in + // MiB. assert_eq!(DEFAULT_CACHE_MB * 1024 * 1024, 1024 * 1024 * 1024); } } diff --git a/crates/crust-assets/src/texture_cache.rs b/crates/crust-assets/src/texture_cache.rs new file mode 100644 index 00000000..95dd951e --- /dev/null +++ b/crates/crust-assets/src/texture_cache.rs @@ -0,0 +1,78 @@ +//! What the two streaming texture caches — the `.tx` [`TileCache`] and the +//! Ptex [`PtexStream`] — share: their budgets' unit, and the per-thread +//! microcache set that keeps a texel fetch off every lock. +//! +//! [`TileCache`]: crate::tiled::TileCache +//! [`PtexStream`]: crate::PtexStream + +/// Bytes in `mib` mebibytes — the unit every `CRUST_*_MB` budget is given in. +pub(crate) const fn mib_to_bytes(mib: u64) -> u64 { + mib * 1024 * 1024 +} + +/// `bytes` in mebibytes, for a log line. +pub(crate) fn bytes_to_mib(bytes: u64) -> f64 { + bytes as f64 / (1024.0 * 1024.0) +} + +/// One set of a per-thread microcache: up to `N` `(key, value)` pairs, newest +/// first. +/// +/// FIFO rather than LRU: promoting on a hit would make every hit a write, and +/// a hit is the common case by far (a bilinear tap reads one tile up to four +/// times in a row). A lookup scans `N` keys, and finds its hit at index 0 the +/// vast majority of the time. +pub(crate) struct Ways([Option<(K, V)>; N]); + +impl Ways { + pub(crate) const EMPTY: Self = Ways([const { None }; N]); + + /// The value held for `key`, if any. + #[inline] + pub(crate) fn get(&self, key: &K) -> Option<&V> { + let idx = self + .0 + .iter() + .position(|s| matches!(s, Some((k, _)) if k == key))?; + self.0[idx].as_ref().map(|(_, v)| v) + } + + /// Puts `(key, value)` in front, returning the oldest entry when it falls + /// off the end — so a caller accounting for what the set holds can + /// subtract it. + #[inline] + pub(crate) fn push(&mut self, key: K, value: V) -> Option<(K, V)> { + let evicted = self.0[N - 1].take(); + self.0.rotate_right(1); + self.0[0] = Some((key, value)); + evicted + } + + /// The values held, newest first. + pub(crate) fn values(&self) -> impl Iterator { + self.0.iter().flatten().map(|(_, v)| v) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn ways_keep_the_newest_n_and_hand_back_the_evicted() { + let mut w: Ways = Ways::EMPTY; + assert!(w.get(&1).is_none()); + assert!(w.push(1, "a").is_none()); + assert!(w.push(2, "b").is_none()); + assert_eq!(w.get(&1), Some(&"a")); + assert_eq!(w.push(3, "c"), Some((1, "a"))); + assert!(w.get(&1).is_none()); + assert_eq!(w.values().copied().collect::>(), ["c", "b"]); + } + + #[test] + fn budgets_are_mebibytes() { + assert_eq!(mib_to_bytes(1), 1 << 20); + assert_eq!(bytes_to_mib(3 << 20), 3.0); + } +} diff --git a/crates/crust-assets/src/tiled/cache.rs b/crates/crust-assets/src/tiled/cache.rs index 5eee3625..273e6bb0 100644 --- a/crates/crust-assets/src/tiled/cache.rs +++ b/crates/crust-assets/src/tiled/cache.rs @@ -30,6 +30,7 @@ //! returns `None` for the caller to turn into a fallback colour. use super::{TileReader, TiledFile}; +use crate::texture_cache::{Ways, mib_to_bytes}; use half::f16; use std::collections::HashMap; use std::sync::atomic::{AtomicU64, Ordering}; @@ -123,7 +124,7 @@ fn thread_stripe() -> usize { } /// Default cache budget, matching OIIO's own 1 GB. -pub const DEFAULT_BUDGET_BYTES: u64 = crust_core::config::DEFAULT_CACHE_MB as u64 * 1024 * 1024; +pub const DEFAULT_BUDGET_BYTES: u64 = mib_to_bytes(crust_core::config::DEFAULT_CACHE_MB as u64); /// Which tile, of which level, of which file. /// @@ -427,21 +428,16 @@ impl TileCache { shards: (0..SHARDS).map(|_| Mutex::new(HashMap::new())).collect(), files: Mutex::new(Vec::new()), resident: AtomicU64::new(0), - budget: budget_bytes.max(1024 * 1024), + budget: budget_bytes.max(mib_to_bytes(1)), sweeping: Mutex::new(0), stats: CacheStats::default(), } } - /// The budget from `CRUST_TEX_CACHE_MB` as parsed into - /// [`crust_core::config()`], or [`DEFAULT_BUDGET_BYTES`]. - pub fn budget_from_env() -> u64 { - Self::budget_of(crust_core::config()) - } - - /// The tile cache budget `config` asks for, in bytes. + /// The tile cache budget `config` asks for (`CRUST_TEX_CACHE_MB`), in + /// bytes. pub fn budget_of(config: &crust_core::Config) -> u64 { - config.tex_cache_mb.get() * 1024 * 1024 + mib_to_bytes(config.tex_cache_mb.get()) } pub fn budget(&self) -> u64 { @@ -720,12 +716,11 @@ const MICRO_SETS: usize = 16; /// a tile edge doubles that. const MICRO_WAYS: usize = 4; -type MicroSlot = Option<((u32, TileId), Arc)>; -const EMPTY_WAYS: [MicroSlot; MICRO_WAYS] = [const { None }; MICRO_WAYS]; +type MicroSet = Ways<(u32, TileId), Arc, MICRO_WAYS>; /// The per-thread microcache: `MICRO_SETS` sets of `MICRO_WAYS` `(key, tile)` /// pairs, newest first within a set, keyed by the owning cache's /// [`TileCache::id`] as well as the tile. -type MicroSlots = [[MicroSlot; MICRO_WAYS]; MICRO_SETS]; +type MicroSlots = [MicroSet; MICRO_SETS]; thread_local! { /// The most recently used tiles, per thread and per texture. @@ -748,7 +743,7 @@ thread_local! { /// MICRO_WAYS` = 64 tiles a thread, 1.5 MiB of `half` 64x64 tiles, so /// 108 MiB on 72 threads against the 1 GiB default (half that for `u8`). static MICRO: std::cell::RefCell = - const { std::cell::RefCell::new([EMPTY_WAYS; MICRO_SETS]) }; + const { std::cell::RefCell::new([MicroSet::EMPTY; MICRO_SETS]) }; } /// The microcache set a key lives in. Files are interned with consecutive @@ -785,11 +780,7 @@ pub fn with_tile(cache: &TileCache, id: TileId, f: impl FnOnce(&Tile) -> R) - let set = micro_set(&key); let hit = MICRO.with(|m| { let slots = m.borrow(); - let ways = &slots[set]; - let idx = ways - .iter() - .position(|s| matches!(s, Some((k, _)) if *k == key))?; - let (_, tile) = ways[idx].as_ref()?; + let tile = slots[set].get(&key)?; Some(f.take()?(tile)) }); if let Some(r) = hit { @@ -800,14 +791,9 @@ pub fn with_tile(cache: &TileCache, id: TileId, f: impl FnOnce(&Tile) -> R) - // decode and must not run under a thread-local borrow. let tile = cache.get(id)?; let r = f.take()?(&tile); - MICRO.with(|m| { - let mut slots = m.borrow_mut(); - // Newest in front; the oldest way falls off the end. FIFO rather than - // LRU: promoting on a hit would make every hit a write. - let ways = &mut slots[set]; - ways.rotate_right(1); - ways[0] = Some((key, tile)); - }); + // Newest in front; the oldest way falls off the end (see `Ways`). + let evicted = MICRO.with(|m| m.borrow_mut()[set].push(key, tile)); + drop(evicted); Some(r) } @@ -817,7 +803,7 @@ pub fn with_tile(cache: &TileCache, id: TileId, f: impl FnOnce(&Tile) -> R) - /// is about to drop, and a stale hit would answer from the wrong cache. #[cfg(test)] pub fn clear_microcache() { - MICRO.with(|m| *m.borrow_mut() = [EMPTY_WAYS; MICRO_SETS]); + MICRO.with(|m| *m.borrow_mut() = [MicroSet::EMPTY; MICRO_SETS]); } #[cfg(test)] diff --git a/crates/crust-assets/src/tiled/stream.rs b/crates/crust-assets/src/tiled/stream.rs index 70581327..82fac4f4 100644 --- a/crates/crust-assets/src/tiled/stream.rs +++ b/crates/crust-assets/src/tiled/stream.rs @@ -87,12 +87,7 @@ impl StreamingTexture { /// /// Returns `None` when nothing could be opened, so the caller can fall /// back to the preload path rather than render an untextured surface. - pub fn open( - path: &Path, - space: ColorSpace, - cache: Arc, - expand: impl Fn(u32, u32) -> Option, - ) -> Option { + pub fn open(path: &Path, space: ColorSpace, cache: Arc) -> Option { let name = path.to_string_lossy().into_owned(); let tiled = name.contains("") || name.contains(""); // `Auto` is settled by the first file that opens (see @@ -103,46 +98,19 @@ impl StreamingTexture { let mut settle = |f: &TiledFile| *resolved.get_or_insert_with(|| resolve_auto_space(f, working)); - let mut charts = Vec::new(); - if tiled { - // The same 10x10 sweep the preload path does, and for the same - // reason: only tiles present on disk cost anything, so a chart - // with holes is free where it has none. - for v in 0..10u32 { - for u in 0..10u32 { - let Some(p) = expand(u, v) else { continue }; - if !p.exists() { - continue; - } - let opened = TiledFile::open(&p).map(|f| { - let want = crate::tiled::space_name(settle(&f)); - (f, want) - }); - match opened { - Ok((f, want_space)) if !f.mip_space_matches(want_space) => { - tracing::warn!( - "{}: mip chain was reduced in {:?}, not {want_space} — \ - falling back to the preloaded texture", - p.display(), - f.mip_space() - ); - return None; - } - Ok((f, _)) => { - if let Some(id) = cache.intern(f.clone()) { - charts.push(Chart { - number: 1001 + u + 10 * v, - file: f, - id, - }); - } - } - Err(e) => tracing::debug!("{}: {e}", p.display()), - } - } - } + // Every tile on disk for a set (the preload path's sweep, so the two + // backends cover the same chart), or the one file as tile 1001. + let files: Vec<(u32, std::path::PathBuf)> = if tiled { + crate::uv_texture::existing_tiles(&name) + .into_iter() + .map(|t| (t.number, t.path)) + .collect() } else { - let opened = TiledFile::open(path).map(|f| { + vec![(udim_number(0, 0), path.to_path_buf())] + }; + let mut charts = Vec::new(); + for (number, p) in files { + let opened = TiledFile::open(&p).map(|f| { let want = crate::tiled::space_name(settle(&f)); (f, want) }); @@ -151,20 +119,21 @@ impl StreamingTexture { tracing::warn!( "{}: mip chain was reduced in {:?}, not {want_space} — \ falling back to the preloaded texture", - path.display(), + p.display(), f.mip_space() ); return None; } Ok((f, _)) => { - let id = cache.intern(f.clone())?; - charts.push(Chart { - number: 1001, - file: f, - id, - }); + if let Some(id) = cache.intern(f.clone()) { + charts.push(Chart { + number, + file: f, + id, + }); + } } - Err(e) => tracing::debug!("{}: {e}", path.display()), + Err(e) => tracing::debug!("{}: {e}", p.display()), } } if charts.is_empty() { @@ -423,8 +392,8 @@ mod tests { let (png, tx) = pair("agree", 256, 192, crust_core::ResolvedColorSpace::SRGB); let pre = UvTexture::open_with(&png, crust_core::ColorSpace::SRGB, true).expect("preload"); let cache = Arc::new(TileCache::new(64 * 1024 * 1024)); - let stream = StreamingTexture::open(&tx, crust_core::ColorSpace::SRGB, cache, |_, _| None) - .expect("stream"); + let stream = + StreamingTexture::open(&tx, crust_core::ColorSpace::SRGB, cache).expect("stream"); assert_eq!(stream.level_count(), pre.level_count()); assert_eq!(stream.size(), pre.tile_size()); @@ -453,8 +422,8 @@ mod tests { let (png, tx) = pair("clipped", 150, 100, crust_core::ResolvedColorSpace::RAW); let pre = UvTexture::open_with(&png, crust_core::ColorSpace::RAW, true).expect("preload"); let cache = Arc::new(TileCache::new(16 * 1024 * 1024)); - let stream = StreamingTexture::open(&tx, crust_core::ColorSpace::RAW, cache, |_, _| None) - .expect("stream"); + let stream = + StreamingTexture::open(&tx, crust_core::ColorSpace::RAW, cache).expect("stream"); // Sampled hard against the right and bottom edges, where the last tile // column is 22 texels wide against a nominal 64. @@ -482,10 +451,7 @@ mod tests { let (png, tx) = pair("thrash", 1024, 1024, crust_core::ResolvedColorSpace::SRGB); let pre = UvTexture::open_with(&png, crust_core::ColorSpace::SRGB, true).expect("preload"); let cache = Arc::new(TileCache::new(1)); - let stream = - StreamingTexture::open(&tx, crust_core::ColorSpace::SRGB, cache.clone(), |_, _| { - None - }) + let stream = StreamingTexture::open(&tx, crust_core::ColorSpace::SRGB, cache.clone()) .expect("stream"); for i in 0..31 { @@ -563,8 +529,8 @@ mod tests { let pre = UvTexture::open_with(&hdr, crust_core::ColorSpace::RAW, true).expect("preload"); let cache = Arc::new(TileCache::new(8 * 1024 * 1024)); - let stream = StreamingTexture::open(&tx, crust_core::ColorSpace::RAW, cache, |_, _| None) - .expect("stream"); + let stream = + StreamingTexture::open(&tx, crust_core::ColorSpace::RAW, cache).expect("stream"); assert!(stream.is_linear(), "an EXR backing pages in half tiles"); // Point-sampled at texel centres, so no interpolation blurs the two @@ -635,15 +601,8 @@ mod tests { TiledFile::open(&tx).expect("open").mip_space(), Some("srgb_texture") ); - assert!( - StreamingTexture::open(&tx, crust_core::ColorSpace::SRGB, cache.clone(), |_, _| { - None - }) - .is_some() - ); - assert!( - StreamingTexture::open(&tx, crust_core::ColorSpace::RAW, cache, |_, _| None).is_none() - ); + assert!(StreamingTexture::open(&tx, crust_core::ColorSpace::SRGB, cache.clone()).is_some()); + assert!(StreamingTexture::open(&tx, crust_core::ColorSpace::RAW, cache).is_none()); let _ = std::fs::remove_dir_all(&dir); } @@ -670,7 +629,7 @@ mod tests { let auto = crust_core::ColorSpace::AUTO.into_working(acescg); let pre = UvTexture::open_with(&exr, auto, false).expect("preload"); let cache = Arc::new(TileCache::new(4 * 1024 * 1024)); - let stream = StreamingTexture::open(&exr, auto, cache, |_, _| None).expect("stream"); + let stream = StreamingTexture::open(&exr, auto, cache).expect("stream"); let gamut = crust_core::ResolvedColorSpace::new(crust_core::color::Space::SRGB_TEXTURE, acescg) .gamut() @@ -722,22 +681,11 @@ mod tests { let cache = Arc::new(TileCache::new(4 * 1024 * 1024)); // Same space: opens. - assert!( - StreamingTexture::open(&tx, crust_core::ColorSpace::SRGB, cache.clone(), |_, _| { - None - }) - .is_some() - ); + assert!(StreamingTexture::open(&tx, crust_core::ColorSpace::SRGB, cache.clone()).is_some()); // Different space: declines rather than serving a chain reduced for // the other one. - assert!( - StreamingTexture::open(&tx, crust_core::ColorSpace::RAW, cache.clone(), |_, _| None) - .is_none() - ); - assert!( - StreamingTexture::open(&tx, crust_core::ColorSpace::GAMMA22, cache, |_, _| None) - .is_none() - ); + assert!(StreamingTexture::open(&tx, crust_core::ColorSpace::RAW, cache.clone()).is_none()); + assert!(StreamingTexture::open(&tx, crust_core::ColorSpace::GAMMA22, cache).is_none()); // And the guard really is load-bearing: had it not fired, the levels // would have differed from what a raw-decoded preload builds. @@ -781,7 +729,6 @@ mod tests { std::path::Path::new("/definitely/not/here.tx"), crust_core::ColorSpace::RAW, cache, - |_, _| None, ) .is_none() ); diff --git a/crates/crust-assets/src/uv_texture/mod.rs b/crates/crust-assets/src/uv_texture/mod.rs index 9cfbdb20..e1b4e61f 100644 --- a/crates/crust-assets/src/uv_texture/mod.rs +++ b/crates/crust-assets/src/uv_texture/mod.rs @@ -70,7 +70,7 @@ pub(crate) use udim::udim_number; use crate::mip_filter::{MipSource, Taps, lerp_rgba, trilinear}; pub(crate) use mip::{reduce_half, reduce_half_linear}; -pub(crate) use udim::expand_token; +pub(crate) use udim::existing_tiles; /// The decoded tiles, in whichever sample type the file warranted. enum Storage { @@ -205,21 +205,10 @@ impl UvTexture { Ok::<_, AssetError>(tile) }; if let Some(token) = token { - // Only tiles that exist on disk are opened, so a chart with holes - // costs nothing for the tiles it does not use. 10x10 covers the - // 1001..1100 range every DCC writes — and is what bounds the - // `` sweep too, since the two tokens name the same grid. - for v in 0..10u32 { - for u in 0..10u32 { - let candidate = token.expand(&name, u, v); - let p = Path::new(&candidate); - if !p.exists() { - continue; - } - match decode(p, udim_number(u, v)) { - Ok(t) => tiles.push(t), - Err(e) => warn!("{e} — skipping that UDIM tile"), - } + for tile in existing_tiles(&name) { + match decode(&tile.path, tile.number) { + Ok(t) => tiles.push(t), + Err(e) => warn!("{e} — skipping that UDIM tile"), } } if tiles.is_empty() { @@ -276,21 +265,10 @@ impl UvTexture { mip: bool, max_edge: NonZeroUsize, ) -> Result { - let name = path.to_string_lossy().into_owned(); - let name = name.as_str(); - let candidates = |token: TileToken| { - (0..10u32) - .flat_map(move |v| (0..10u32).map(move |u| (u, v))) - .filter_map(move |(u, v)| { - let candidate = token.expand(name, u, v); - Path::new(&candidate) - .exists() - .then(|| (u, v, std::path::PathBuf::from(candidate))) - }) - }; + let found = existing_tiles(&path.to_string_lossy()); // Settled by the first file, as `StreamingTexture::open` settles it. let first = match token { - Some(token) => candidates(token).next().map(|(_, _, p)| p), + Some(_) => found.first().map(|t| t.path.clone()), None => Some(path.to_path_buf()), }; let marked = match space.resolved() { @@ -310,8 +288,8 @@ impl UvTexture { }; let mut tiles = Vec::new(); if let Some(token) = token { - for (u, v, p) in candidates(token) { - match decode_exr_tile(&p, udim_number(u, v), max_edge, decode) { + for tile in &found { + match decode_exr_tile(&tile.path, tile.number, max_edge, decode) { Ok(t) => tiles.push(t), Err(e) => warn!("{e} — skipping that UDIM tile"), } diff --git a/crates/crust-assets/src/uv_texture/udim.rs b/crates/crust-assets/src/uv_texture/udim.rs index de353c98..e2638f26 100644 --- a/crates/crust-assets/src/uv_texture/udim.rs +++ b/crates/crust-assets/src/uv_texture/udim.rs @@ -1,6 +1,8 @@ //! Tile-set addressing: the `` / `` filename tokens and the //! UDIM numbering every tile is keyed by. +use std::path::PathBuf; + /// The filename token that addresses a tile set, and how it spells a tile. /// /// Two spellings, one grid: a document may use either, and both index the @@ -48,14 +50,37 @@ impl TileToken { } } -/// Expands a `` / `` token in `name` for chart coordinates -/// `(u, v)`, or `None` when the name carries no token. +/// One tile of a `` / `` set that exists on disk. +pub(crate) struct TileFile { + /// Its UDIM number ([`udim_number`]), whichever token named the file. + pub(crate) number: u32, + pub(crate) path: PathBuf, +} + +/// Every tile of the set `name` addresses that exists on disk, in UDIM order +/// (row by row) — or nothing when `name` carries no token. /// -/// Shared with the streaming path so the two discover the same set of tiles: -/// a sweep that disagreed about which files exist would make the two texture -/// backends cover different parts of the chart. -pub(crate) fn expand_token(name: &str, u: u32, v: u32) -> Option { - TileToken::detect(name).map(|t| t.expand(name, u, v)) +/// The one sweep every reader of a tile set takes — the preload path, the +/// streaming path, the `.tx` conversion — so they discover the same set: a +/// sweep that disagreed about which files exist would make two texture +/// backends cover different parts of the chart. Only tiles present on disk +/// are returned, so a chart with holes costs nothing for the tiles it does +/// not use. 10x10 covers the 1001..1100 range every DCC writes, and is what +/// bounds the `` sweep too, since the two tokens name the same grid. +pub(crate) fn existing_tiles(name: &str) -> Vec { + let Some(token) = TileToken::detect(name) else { + return Vec::new(); + }; + (0..10u32) + .flat_map(|v| (0..10u32).map(move |u| (u, v))) + .filter_map(|(u, v)| { + let path = PathBuf::from(token.expand(name, u, v)); + path.exists().then(|| TileFile { + number: udim_number(u, v), + path, + }) + }) + .collect() } /// The UDIM number of the tile at zero-based chart coordinates. diff --git a/crates/crust-core/src/scene/usd_import/assets.rs b/crates/crust-core/src/scene/usd_import/assets.rs new file mode 100644 index 00000000..8ed931b0 --- /dev/null +++ b/crates/crust-core/src/scene/usd_import/assets.rs @@ -0,0 +1,169 @@ +//! Asset paths and host decodes: the one rule that turns an `asset`-valued +//! attribute into a filesystem path ([`asset_path`]), and the one memoized, +//! timed route every decode takes through the host's +//! [`AssetLoader`](crate::scene::AssetLoader) ([`cached_asset`]). +//! crust-core decodes nothing itself. + +use std::collections::HashMap; +use std::hash::Hash; +use std::path::{Path, PathBuf}; +use std::sync::Arc; +use std::time::{Duration, Instant}; + +use openusd::sdf; +use openusd::usd::Attribute; +use tracing::debug; + +use super::ImportCaches; +use super::attrs::value_at; + +/// Runs one host decode, adding its wall time to `asset_time` — the "Load +/// assets" phase, which the traversal figure has subtracted out (see +/// `ImportCaches::asset_time`). +pub(super) fn timed_asset(asset_time: &mut Duration, load: impl FnOnce() -> V) -> V { + let started = Instant::now(); + let loaded = load(); + *asset_time += started.elapsed(); + loaded +} + +/// One host decode, memoized by `key` and timed by [`timed_asset`]. +/// +/// Negative results are cached too: a file that failed to open (a missing +/// texture, a 600 MB Ptex the host declined) is not retried per material or +/// light. `load` runs only on a miss, so whatever it reports is reported once +/// per key. +pub(super) fn cached_asset( + cache: &mut HashMap>, + asset_time: &mut Duration, + key: K, + load: impl FnOnce(&K) -> Option, +) -> Option { + if let Some(hit) = cache.get(&key) { + return hit.clone(); + } + let loaded = timed_asset(asset_time, || load(&key)); + cache.insert(key, loaded.clone()); + loaded +} + +/// Opens a UV texture through the host, memoized by resolved path and colour +/// space. The caller maps its own vocabulary onto the space — MaterialX's +/// `colorspace` through [`crate::ColorSpace::from_mtlx`], UsdUVTexture's +/// `sourceColorSpace` through [`crate::ColorSpace::from_usd`] — since the two +/// disagree on what an absent attribute means. +pub(super) fn load_uv_texture( + path: &Path, + space: crate::ColorSpace, + caches: &mut ImportCaches<'_>, +) -> Option> { + let key = (path.to_string_lossy().into_owned(), space); + let assets = caches.assets; + cached_asset( + &mut caches.materials.textures, + &mut caches.asset_time, + key, + |_| { + let loaded = assets.load_texture(path, space); + if loaded.is_none() { + debug!( + "Texture {} ({space:?}) not loadable by the host", + path.display() + ); + } + loaded + }, + ) +} + +/// Opens a Ptex file through the host, once per `(resolved path, space)`. +/// +/// Keyed on the resolved filesystem path, which — unlike a prototype-scoped +/// scene path — is stable across the streaming importer's stages, so one +/// texture is opened once however many materials or chunks reference it. +/// The space is in the key because the decode happens at open: a file read +/// both as colour and as displacement is two textures, which is correct and +/// rare. +pub(super) fn load_ptex( + path: &Path, + space: crate::ColorSpace, + caches: &mut ImportCaches<'_>, +) -> Option> { + let key = (path.to_string_lossy().into_owned(), space); + let assets = caches.assets; + cached_asset( + &mut caches.materials.ptex, + &mut caches.asset_time, + key, + |_| assets.load_ptex(path, space), + ) +} + +/// An `asset`-valued attribute as a filesystem path — the one rule every +/// asset reference (textures, Ptex, IES profiles, light and dome maps) is +/// resolved by. +/// +/// openusd anchors default-sourced asset paths against the layer that +/// authored them and reports the result in `resolved_path` — which is what +/// makes a production stage's `../../../textures/foo.ptx` work at all, since +/// the layer authoring it is nested several directories below the root (the +/// Moana island's lights author `../textures/islandsun.exr` relative to +/// `usd/island.usda`). +/// +/// But openusd reports the anchored path only when it names a file that +/// exists — and a ``-tokened texture path never does, since it names a +/// set. So such a value arrives with no `resolved_path` at all, and an +/// **unresolved** relative path is anchored here against the layer that +/// authored it: the strongest spec in the attribute's property stack, the +/// layer whose opinion supplies the value, which is exactly what USD anchors +/// against. Anchoring it against the root layer was wrong for any texture +/// authored in a sublayer or reference: ALab's look layers sit five +/// directories below `entry.usda` and author `@../../texture/….exr@`, +/// and 1 718 texture sets failed to load. The root layer (`stage_path`) is +/// only the last resort, when the authoring layer is unknown. +pub(super) fn asset_path(attr: &Attribute, stage_path: &Path) -> Option { + let value = value_at(attr)?; + let resolved = matches!(&value, sdf::Value::AssetPath(p) + if p.resolved_path().is_some_and(|r| !r.is_empty())); + let authored = value.as_str().map(str::to_owned); + if !resolved + && let Some(authored) = authored.filter(|a| !a.is_empty() && Path::new(a).is_relative()) + && let Some(layer_dir) = attr + .property_stack() + .ok() + .and_then(|stack| stack.into_iter().next()) + .and_then(|site| Path::new(&site.layer).parent().map(Path::to_path_buf)) + .filter(|d| !d.as_os_str().is_empty()) + { + return Some(layer_dir.join(authored)); + } + asset_value_path(&value, stage_path) +} + +/// An asset value as a path: openusd's `resolved_path` when it has one, else +/// the authored string, a relative one anchored against the root layer. +fn asset_value_path(value: &sdf::Value, stage_path: &Path) -> Option { + let (authored, resolved) = match value { + sdf::Value::AssetPath(p) => (p.as_str().to_string(), p.resolved_path()), + sdf::Value::String(p) => (p.clone(), None), + _ => return None, + }; + if let Some(r) = resolved + && !r.is_empty() + { + return Some(PathBuf::from(r)); + } + if authored.is_empty() { + return None; + } + let candidate = Path::new(&authored); + if candidate.is_absolute() { + return Some(candidate.to_path_buf()); + } + Some( + stage_path + .parent() + .unwrap_or_else(|| Path::new(".")) + .join(candidate), + ) +} diff --git a/crates/crust-core/src/scene/usd_import/attrs.rs b/crates/crust-core/src/scene/usd_import/attrs.rs index 77dc268f..fd872860 100644 --- a/crates/crust-core/src/scene/usd_import/attrs.rs +++ b/crates/crust-core/src/scene/usd_import/attrs.rs @@ -3,8 +3,9 @@ //! schema-attribute value decoding. Every read resolves at [`eval_time`]. use glam::{Vec3, Vec3A}; +use openusd::gf::Vec3f; use openusd::sdf; -use openusd::usd::Prim; +use openusd::usd::{Attribute, Prim}; use tracing::warn; use crate::color::Space; @@ -151,34 +152,53 @@ pub(super) fn resolve_adaptive_max_level(host: Option, authored: Option Option { - let v = prim - .attribute(name) - .get_at::(eval_time()) - .ok()??; +// ----------------------------------------------------------------------- +// Value reads and decoders +// ----------------------------------------------------------------------- +// +// Every attribute read in the importer goes through [`value_at`] and one of +// the `decode_*` functions below, so that two readers of the same kind of +// value cannot disagree about which authored types they accept. + +/// `attr`'s value at [`eval_time`], or `None` when it is unauthored, blocked +/// or unreadable. +pub(super) fn value_at(attr: &Attribute) -> Option { + attr.get_at::(eval_time()).ok().flatten() +} + +/// The value of `prim`'s attribute `name` at [`eval_time`] — [`value_at`] by +/// name. +pub(super) fn prim_value(prim: &Prim, name: &str) -> Option { + value_at(&prim.attribute(name)) +} + +/// A scalar float, whatever precision it was authored in. +pub(super) fn decode_f32(v: sdf::Value) -> Option { match v { - sdf::Value::Int(i) => Some(i), + sdf::Value::Float(f) => Some(f), + sdf::Value::Double(d) => Some(d as f32), + sdf::Value::Half(h) => Some(h.to_f32()), _ => None, } } -pub(super) fn custom_f32(prim: &Prim, name: &str) -> Option { - let v = prim - .attribute(name) - .get_at::(eval_time()) - .ok()??; +/// A number where hand-written files author integers as often as floats +/// (`clearValue`, MaterialX-style shader inputs): [`decode_f32`] plus `int`. +pub(super) fn decode_number(v: sdf::Value) -> Option { match v { - sdf::Value::Float(f) => Some(f), - sdf::Value::Double(d) => Some(d as f32), + sdf::Value::Int(i) => Some(i as f32), + v => decode_f32(v), + } +} + +pub(super) fn decode_i32(v: sdf::Value) -> Option { + match v { + sdf::Value::Int(i) => Some(i), _ => None, } } -pub(super) fn custom_bool(prim: &Prim, name: &str) -> Option { - let v = prim - .attribute(name) - .get_at::(eval_time()) - .ok()??; +pub(super) fn decode_bool(v: sdf::Value) -> Option { match v { sdf::Value::Bool(b) => Some(b), // Authoring tools sometimes write bools as ints. @@ -187,11 +207,8 @@ pub(super) fn custom_bool(prim: &Prim, name: &str) -> Option { } } -pub(super) fn custom_token(prim: &Prim, name: &str) -> Option { - let v = prim - .attribute(name) - .get_at::(eval_time()) - .ok()??; +/// A token or a string — exporters write either for the same attribute. +pub(super) fn decode_text(v: sdf::Value) -> Option { match v { sdf::Value::Token(t) => Some(t.as_str().to_owned()), sdf::Value::String(s) => Some(s), @@ -199,18 +216,83 @@ pub(super) fn custom_token(prim: &Prim, name: &str) -> Option { } } -pub(super) fn custom_color3(prim: &Prim, name: &str) -> Option { - let v = prim - .attribute(name) - .get_at::(eval_time()) - .ok()??; +/// A three-vector (`color3f`, `float3`, `vector3d`, …) in any precision. +/// USD has no dedicated colour variant: `color3f` is a `Vec3f`. +pub(super) fn decode_vec3(v: sdf::Value) -> Option { match v { sdf::Value::Vec3f(c) => Some(Vec3A::new(c.x, c.y, c.z)), sdf::Value::Vec3d(c) => Some(Vec3A::new(c.x as f32, c.y as f32, c.z as f32)), + sdf::Value::Vec3h(c) => Some(Vec3A::new(c.x.to_f32(), c.y.to_f32(), c.z.to_f32())), + _ => None, + } +} + +/// A shading value widened to four channels, the shape `UsdUVTexture`'s +/// `scale`/`bias`/`fallback` have: a `float4` as authored, a colour with +/// alpha 1, a scalar in every channel. +pub(super) fn decode_float4(v: sdf::Value) -> Option<[f32; 4]> { + match v { + sdf::Value::Vec4f(v) => Some([v.x, v.y, v.z, v.w]), + sdf::Value::Vec4d(v) => Some([v.x as f32, v.y as f32, v.z as f32, v.w as f32]), + sdf::Value::Vec4h(v) => Some([v.x.to_f32(), v.y.to_f32(), v.z.to_f32(), v.w.to_f32()]), + v @ (sdf::Value::Vec3f(_) | sdf::Value::Vec3d(_) | sdf::Value::Vec3h(_)) => { + decode_vec3(v).map(|c| c.extend(1.0).to_array()) + } + v => decode_f32(v).map(|f| [f; 4]), + } +} + +pub(super) fn decode_f32_array(v: sdf::Value) -> Option> { + match v { + sdf::Value::FloatVec(v) => Some(v), + sdf::Value::DoubleVec(v) => Some(v.into_iter().map(|d| d as f32).collect()), + sdf::Value::HalfVec(v) => Some(v.into_iter().map(|h| h.to_f32()).collect()), + _ => None, + } +} + +pub(super) fn decode_i32_array(v: sdf::Value) -> Option> { + match v { + sdf::Value::IntVec(v) => Some(v), + _ => None, + } +} + +pub(super) fn decode_i64_array(v: sdf::Value) -> Option> { + match v { + sdf::Value::Int64Vec(v) => Some(v), + _ => None, + } +} + +/// A `point3f[]` / `float3[]` array as authored. +pub(super) fn decode_vec3f_array(v: sdf::Value) -> Option> { + match v { + sdf::Value::Vec3fVec(v) => Some(v), _ => None, } } +pub(super) fn custom_i32(prim: &Prim, name: &str) -> Option { + prim_value(prim, name).and_then(decode_i32) +} + +pub(super) fn custom_f32(prim: &Prim, name: &str) -> Option { + prim_value(prim, name).and_then(decode_f32) +} + +pub(super) fn custom_bool(prim: &Prim, name: &str) -> Option { + prim_value(prim, name).and_then(decode_bool) +} + +pub(super) fn custom_token(prim: &Prim, name: &str) -> Option { + prim_value(prim, name).and_then(decode_text) +} + +pub(super) fn custom_color3(prim: &Prim, name: &str) -> Option { + prim_value(prim, name).and_then(decode_vec3) +} + /// A colour attribute in the working space: [`custom_color3`], converted /// from the space its `colorSpace` metadatum names (see [`in_working`]). pub(super) fn custom_color(prim: &Prim, name: &str, working: Space) -> Option { @@ -225,17 +307,13 @@ pub(super) fn custom_color(prim: &Prim, name: &str, working: Space) -> Option Option { +pub(super) fn attr_color_space(attr: &Attribute) -> Option { let name = own_color_space_name(attr).or_else(|| { let stage = attr.stage(); let mut path = Some(attr.path().prim_path()); while let Some(p) = path.filter(|p| !p.is_abs_root()) { - let named = super::prim_at(stage, p.clone()) - .attribute("colorSpace:name") - .get_at::(eval_time()) - .ok() - .flatten() - .and_then(token_text) + let named = prim_value(&super::prim_at(stage, p.clone()), "colorSpace:name") + .and_then(decode_text) .filter(|n| !n.is_empty()); if named.is_some() { return named; @@ -255,29 +333,21 @@ pub(super) fn attr_color_space(attr: &openusd::usd::Attribute) -> Option /// A scope's `colorSpace:name = "acescg"` says its *linear colour values* are /// ACEScg; applied to an sRGB albedo file it would read it as linear, and to /// a displacement map it would mix its channels. -pub(super) fn attr_own_color_space(attr: &openusd::usd::Attribute) -> Option { +pub(super) fn attr_own_color_space(attr: &Attribute) -> Option { let name = own_color_space_name(attr)?; named_space(attr, &name) } -fn token_text(v: sdf::Value) -> Option { - match v { - sdf::Value::Token(t) => Some(t.as_str().to_owned()), - sdf::Value::String(s) => Some(s), - _ => None, - } -} - -fn own_color_space_name(attr: &openusd::usd::Attribute) -> Option { +fn own_color_space_name(attr: &Attribute) -> Option { attr.get_metadata::("colorSpace") .ok() .flatten() - .and_then(token_text) + .and_then(decode_text) .filter(|n| !n.is_empty()) } /// `name` through the OCIO config, warning when it does not know it. -fn named_space(attr: &openusd::usd::Attribute, name: &str) -> Option { +fn named_space(attr: &Attribute, name: &str) -> Option { let space = Space::named(name); if space.is_none() { warn!( @@ -294,7 +364,7 @@ fn named_space(attr: &openusd::usd::Attribute, name: &str) -> Option { /// none is taken as already in the working space — UsdLux's "in the rendering /// color space", and the rule every unmanaged input follows /// ([`crate::color`]). -pub(super) fn in_working(attr: &openusd::usd::Attribute, rgb: Vec3A, working: Space) -> Vec3A { +pub(super) fn in_working(attr: &Attribute, rgb: Vec3A, working: Space) -> Vec3A { match attr_color_space(attr) { Some(space) => crate::color::convert(rgb, space, working), None => rgb, @@ -302,55 +372,23 @@ pub(super) fn in_working(attr: &openusd::usd::Attribute, rgb: Vec3A, working: Sp } pub(super) fn custom_f32_array(prim: &Prim, name: &str) -> Option> { - let v = prim - .attribute(name) - .get_at::(eval_time()) - .ok()??; - match v { - sdf::Value::FloatVec(v) => Some(v), - sdf::Value::DoubleVec(v) => Some(v.into_iter().map(|d| d as f32).collect()), - _ => None, - } + prim_value(prim, name).and_then(decode_f32_array) } pub(super) fn custom_i32_array(prim: &Prim, name: &str) -> Option> { - let v = prim - .attribute(name) - .get_at::(eval_time()) - .ok()??; - match v { - sdf::Value::IntVec(v) => Some(v), - _ => None, - } + prim_value(prim, name).and_then(decode_i32_array) } -// ----------------------------------------------------------------------- -// Attribute helpers -// ----------------------------------------------------------------------- - -pub(super) fn attr_f32(attr: &openusd::usd::Attribute) -> Option { - match attr.get_at::(eval_time()).ok()?? { - sdf::Value::Float(f) => Some(f), - sdf::Value::Double(d) => Some(d as f32), - _ => None, - } +pub(super) fn attr_f32(attr: &Attribute) -> Option { + value_at(attr).and_then(decode_f32) } -pub(super) fn attr_bool(attr: &openusd::usd::Attribute) -> Option { - match attr.get_at::(eval_time()).ok()?? { - sdf::Value::Bool(b) => Some(b), - // Authoring tools sometimes write bools as ints. - sdf::Value::Int(i) => Some(i != 0), - _ => None, - } +pub(super) fn attr_bool(attr: &Attribute) -> Option { + value_at(attr).and_then(decode_bool) } -pub(super) fn attr_color3f(attr: &openusd::usd::Attribute) -> Option<[f32; 3]> { - match attr.get_at::(eval_time()).ok()?? { - // color3f is stored as Vec3f in sdf::Value - sdf::Value::Vec3f(v) => Some([v.x, v.y, v.z]), - _ => None, - } +pub(super) fn attr_vec3(attr: &Attribute) -> Option { + value_at(attr).and_then(decode_vec3) } #[cfg(test)] diff --git a/crates/crust-core/src/scene/usd_import/camera.rs b/crates/crust-core/src/scene/usd_import/camera.rs index 2b1acc30..4db6c9ad 100644 --- a/crates/crust-core/src/scene/usd_import/camera.rs +++ b/crates/crust-core/src/scene/usd_import/camera.rs @@ -10,7 +10,57 @@ use crate::tracer::RenderSettings; use super::attrs::attr_f32; use super::prim_at; -use super::xform::{local_matrix_at, resets_xform_stack_at}; +use super::xform::compose_with_parent; + +/// One `UsdGeomCamera` as both its readers see it: [`build_camera`], which +/// builds the render camera, and [`screen_projection`], which adaptive +/// subdivision reads before the traversal. Both derive from this one read, so +/// they cannot describe two different cameras. +struct CameraFrame { + /// The camera's position and its view direction and up, in world space. + /// USD's camera looks down local −Z with +Y up. + eye: Vec3, + forward: Vec3, + up: Vec3, + /// The focal length and the vertical aperture, in the same units: the + /// vertical aperture defaults to the horizontal one over the image's + /// aspect ratio. + focal_length: f32, + vert_aperture: f32, + /// The image size the aperture default and the field of view assume. + width: f32, + height: f32, +} + +impl CameraFrame { + fn read(cam: &UsdCamera, stage: &Stage, prim: &Prim, settings: &RenderSettings) -> Self { + let world = local_to_world(stage, prim); + let focal_length = attr_f32(&cam.focal_length_attr()).unwrap_or(50.0); + let horiz_aperture = attr_f32(&cam.horizontal_aperture_attr()).unwrap_or(20.955); + let (w, h) = settings.get_dimensions(); + let (width, height) = (w as f32, h as f32); + let vert_aperture = + attr_f32(&cam.vertical_aperture_attr()).unwrap_or(horiz_aperture * height / width); + CameraFrame { + eye: world.transform_point3(Vec3::ZERO), + forward: world.transform_vector3(Vec3::NEG_Z).normalize(), + up: world.transform_vector3(Vec3::Y).normalize(), + focal_length, + vert_aperture, + width, + height, + } + } + + /// Tangent of the half vertical field of view. + fn tan_half_vfov(&self) -> f32 { + self.vert_aperture / (2.0 * self.focal_length) + } + + fn aspect(&self) -> f32 { + self.width / self.height + } +} pub(super) fn build_camera( stage: &Stage, @@ -18,32 +68,20 @@ pub(super) fn build_camera( settings: &RenderSettings, ) -> Option { let cam = UsdCamera::get(stage, prim.path().clone()).ok().flatten()?; - let world = local_to_world(stage, prim); - - // USD camera looks down -Z with +Y up in local space. - let lookfrom_v = world.transform_point3(Vec3::ZERO); - let forward_v = world.transform_vector3(Vec3::NEG_Z).normalize(); - let up_v = world.transform_vector3(Vec3::Y).normalize(); - - let (focal_length, vert_aperture) = lens(&cam, settings); + let frame = CameraFrame::read(&cam, stage, prim, settings); let f_stop = attr_f32(&cam.f_stop_attr()).unwrap_or(0.0); let focus_distance = attr_f32(&cam.focus_distance_attr()).unwrap_or(10.0); - let (w, h) = settings.get_dimensions(); - let (w_f, h_f) = (w as f32, h as f32); - - let vfov_deg = 2.0 * (vert_aperture / (2.0 * focal_length)).atan().to_degrees(); + let vfov_deg = 2.0 * frame.tan_half_vfov().atan().to_degrees(); let aperture = if f_stop > 0.0 { - focal_length / f_stop + frame.focal_length / f_stop } else { 0.0 }; - - let aspect = w_f / h_f; - let lookfrom = Vec3A::new(lookfrom_v.x, lookfrom_v.y, lookfrom_v.z); - let lookat_v = lookfrom_v + forward_v * focus_distance; - let lookat = Vec3A::new(lookat_v.x, lookat_v.y, lookat_v.z); - let vup = Vec3A::new(up_v.x, up_v.y, up_v.z); + let aspect = frame.aspect(); + let lookfrom = Vec3A::from(frame.eye); + let lookat = Vec3A::from(frame.eye + frame.forward * focus_distance); + let vup = Vec3A::from(frame.up); debug!( "USD camera: lookfrom={:?} lookat={:?} vup={:?} vfov={} aspect={} aperture={} focus={}", @@ -61,45 +99,29 @@ pub(super) fn build_camera( )) } -/// The focal length and the vertical aperture, in the same units: the vertical -/// aperture defaults to the horizontal one over the image's aspect ratio. -fn lens(cam: &UsdCamera, settings: &RenderSettings) -> (f32, f32) { - let focal_length = attr_f32(&cam.focal_length_attr()).unwrap_or(50.0); - let horiz_aperture = attr_f32(&cam.horizontal_aperture_attr()).unwrap_or(20.955); - let (w, h) = settings.get_dimensions(); - let vert_aperture = - attr_f32(&cam.vertical_aperture_attr()).unwrap_or(horiz_aperture * h as f32 / w as f32); - (focal_length, vert_aperture) -} - /// What adaptive subdivision needs of the render camera, read before the /// traversal builds it: the position and the pixels per world unit at unit /// distance, `image height / (2 tan(vfov / 2))` = `height · focal / aperture`. -/// From the same attributes and the same [`lens`] as [`build_camera`], so the -/// two describe one camera. `None` when `prim` is not a camera on `stage`. +/// The same [`CameraFrame`] as [`build_camera`]'s. `None` when `prim` is not a +/// camera on `stage`. pub(super) fn screen_projection( stage: &Stage, prim: &Prim, settings: &RenderSettings, ) -> Option { let cam = UsdCamera::get(stage, prim.path().clone()).ok().flatten()?; - let world = local_to_world(stage, prim); - let eye = world.transform_point3(Vec3::ZERO); - let (focal_length, vert_aperture) = lens(&cam, settings); - let (w, h) = settings.get_dimensions(); - let f_px = h as f32 * focal_length / vert_aperture; + let frame = CameraFrame::read(&cam, stage, prim, settings); + let f_px = frame.height * frame.focal_length / frame.vert_aperture; // The view pyramid, as `build_camera` builds it: the vertical field of // view from the lens, the horizontal one from the image's aspect ratio. - let tan_v = vert_aperture / (2.0 * focal_length); - let forward = world.transform_vector3(Vec3::NEG_Z).normalize(); - let up = world.transform_vector3(Vec3::Y).normalize(); + let tan_v = frame.tan_half_vfov(); (f_px.is_finite() && f_px > 0.0).then_some(ScreenProjection { - eye, + eye: frame.eye, f_px, - forward, - up, + forward: frame.forward, + up: frame.up, tan_v, - tan_h: tan_v * w as f32 / h as f32, + tan_h: tan_v * frame.width / frame.height, }) } @@ -130,11 +152,7 @@ fn local_to_world(stage: &Stage, prim: &Prim) -> GMat4 { } } ancestors.reverse(); - let mut acc = GMat4::IDENTITY; - for p in &ancestors { - let local = local_matrix_at(stage, p); - let resets = resets_xform_stack_at(stage, p); - acc = if resets { local } else { acc * local }; - } - acc + ancestors + .iter() + .fold(GMat4::IDENTITY, |acc, p| compose_with_parent(stage, p, acc)) } diff --git a/crates/crust-core/src/scene/usd_import/instancing.rs b/crates/crust-core/src/scene/usd_import/instancing.rs index d047ef34..1f294158 100644 --- a/crates/crust-core/src/scene/usd_import/instancing.rs +++ b/crates/crust-core/src/scene/usd_import/instancing.rs @@ -25,7 +25,6 @@ use crust_rt::{ Geometry, InstanceHitId, RayMask, Scene as RtScene, SceneBuilder as RtSceneBuilder, }; use glam::{Affine3A, Mat4 as GMat4, Vec3, Vec3A}; -use openusd::gf::Vec3f; use openusd::sdf; use openusd::usd::{Prim, Stage}; use openusd_schemas::geom::{ @@ -37,13 +36,14 @@ use tracing::{debug, warn}; use crate::material::Material; use crate::rt_world::{FaceMap, UvMap, WorldBuilder}; -use super::attrs::{custom_token, prim_ray_mask}; +use super::attrs::{ + custom_token, decode_i32_array, decode_i64_array, decode_vec3f_array, prim_ray_mask, value_at, +}; use super::materials::{resolve_bound, resolve_material}; use super::mesh::{MeshPlace, mesh_source, placement_scale}; use super::shapes::{curve_segments, sphere_radius}; -use super::time::eval_time; -use super::xform::{local_matrix_at, resets_xform_stack_at}; -use super::{ImportCaches, is_invisible, non_render_purpose, prim_at}; +use super::xform::compose_with_parent; +use super::{ImportCaches, WalkScope, prim_at, prune_reason}; /// How deep prototypes may nest before the importer gives up. USD forbids /// an instancing cycle, but a malformed stage can still describe one, and @@ -309,41 +309,19 @@ pub(super) fn collect_proto_parts( } /// Whether the prototype walk leaves `prim` and its subtree out: the same -/// pruning as the top-level traversal (inactive, a non-render purpose, -/// invisible), plus a native instance nested inside the prototype, which +/// pruning as the top-level traversal ([`prune_reason`], abstractness +/// aside), plus a native instance nested inside the prototype, which /// openusd cannot read. `report` logs why; the placement count's walk passes `false`, /// so a skipped prim is reported once. fn prototype_prunes(prim: &Prim, root: &Prim, report: bool) -> bool { - // Same pruning as the top-level traversal: an inactive prim (and - // its subtree) is absent from the composed scene, prototype or not. - if !prim.is_active().unwrap_or(true) { - if report { - debug!( - "Skipping inactive prim {} (prototype {})", - prim.path(), - root.path() - ); - } - return true; - } - if let Some(purpose) = non_render_purpose(prim) { - if report { - debug!( - "Skipping {purpose}-purpose prim {} (prototype {})", - prim.path(), - root.path() - ); - } - return true; - } // Visibility counts from the prototype root down, as UsdImaging // computes it for a prototype: an invisible part of a prototype is // missing from every instance. No camera is taken from a prototype, - // so here the subtree is simply pruned. - if is_invisible(prim) { + // so here an invisible subtree is simply pruned. + if let Some(reason) = prune_reason(prim, WalkScope::Prototype) { if report { debug!( - "Skipping invisible prim {} (prototype {})", + "Skipping {reason} prim {} (prototype {})", prim.path(), root.path() ); @@ -391,10 +369,8 @@ fn prototype_prunes(prim: &Prim, root: &Prim, report: bool) -> bool { fn part_local(stage: &Stage, prim: &Prim, root: &Prim, parent_local: GMat4) -> GMat4 { if prim.path() == root.path() { GMat4::IDENTITY - } else if resets_xform_stack_at(stage, prim) { - local_matrix_at(stage, prim) } else { - parent_local * local_matrix_at(stage, prim) + compose_with_parent(stage, prim, parent_local) } } @@ -802,9 +778,7 @@ fn read_instancer( } }; - let Ok(Some(sdf::Value::IntVec(proto_indices))) = instancer - .proto_indices_attr() - .get_at::(eval_time()) + let Some(proto_indices) = value_at(&instancer.proto_indices_attr()).and_then(decode_i32_array) else { if report { warn!( @@ -815,20 +789,16 @@ fn read_instancer( return None; }; - let positions = value_vec3f_array(&instancer.positions_attr()).unwrap_or_default(); - let scales = value_vec3f_array(&instancer.scales_attr()); + let positions = value_at(&instancer.positions_attr()) + .and_then(decode_vec3f_array) + .unwrap_or_default(); + let scales = value_at(&instancer.scales_attr()).and_then(decode_vec3f_array); let orientations = instance_orientations(instancer); - let ids = match instancer.ids_attr().get_at::(eval_time()) { - Ok(Some(sdf::Value::Int64Vec(v))) => Some(v), - _ => None, - }; - let invisible: std::collections::HashSet = match instancer - .invisible_ids_attr() - .get_at::(eval_time()) - { - Ok(Some(sdf::Value::Int64Vec(v))) => v.into_iter().collect(), - _ => Default::default(), - }; + let ids = value_at(&instancer.ids_attr()).and_then(decode_i64_array); + let invisible: std::collections::HashSet = value_at(&instancer.invisible_ids_attr()) + .and_then(decode_i64_array) + .map(|v| v.into_iter().collect()) + .unwrap_or_default(); if positions.len() < proto_indices.len() && report { warn!( @@ -1102,29 +1072,15 @@ pub(super) fn emit_point_instancer( ); } -/// A `point3f[]` / `float3[]` attribute as a plain vector. -fn value_vec3f_array(attr: &openusd::usd::Attribute) -> Option> { - match attr.get_at::(eval_time()) { - Ok(Some(sdf::Value::Vec3fVec(v))) => Some(v), - _ => None, - } -} - /// Per-instance rotations, preferring single-precision `orientationsf` /// over half-precision `orientations` as USD specifies. fn instance_orientations(instancer: &PointInstancer) -> Option> { let quat = |w: f32, x: f32, y: f32, z: f32| glam::Quat::from_xyzw(x, y, z, w).normalize(); - if let Ok(Some(sdf::Value::QuatfVec(v))) = instancer - .orientationsf_attr() - .get_at::(eval_time()) - { + if let Some(sdf::Value::QuatfVec(v)) = value_at(&instancer.orientationsf_attr()) { return Some(v.iter().map(|q| quat(q.w, q.x, q.y, q.z)).collect()); } - match instancer - .orientations_attr() - .get_at::(eval_time()) - { - Ok(Some(sdf::Value::QuathVec(v))) => Some( + match value_at(&instancer.orientations_attr()) { + Some(sdf::Value::QuathVec(v)) => Some( v.iter() .map(|q| quat(q.w.to_f32(), q.x.to_f32(), q.y.to_f32(), q.z.to_f32())) .collect(), diff --git a/crates/crust-core/src/scene/usd_import/light_links.rs b/crates/crust-core/src/scene/usd_import/light_links.rs index cb80a6b0..b1bd1b9a 100644 --- a/crates/crust-core/src/scene/usd_import/light_links.rs +++ b/crates/crust-core/src/scene/usd_import/light_links.rs @@ -41,9 +41,8 @@ use crate::ray::{MASK_CAMERA, MASK_SHADOW, RayMask}; use crate::rt_world::WorldBuilder; use crate::volume::VolumeRegion; -use super::attrs::custom_bool; +use super::attrs::{custom_bool, custom_token, prim_value}; use super::prim_at; -use super::time::eval_time; /// The mask bits that carry shadow classes (design D3): 3–30 for the most /// populated classes, 31 shared by the rest. @@ -62,20 +61,13 @@ const ALLOCATED: usize = 28; /// when `includeRoot` is not authored the pseudo-root is added to the rule map /// here, unless an opinion on `/` is already there. pub(super) fn link_query(stage: &Stage, prim: &Prim, name: &str) -> Option { - let prop = |suffix: &str| format!("collection:{name}:{suffix}"); - let targets = |suffix: &str| { - prim.relationship(prop(suffix)) - .targets() - .unwrap_or_default() - }; - let (includes, excludes) = (targets("includes"), targets("excludes")); - let include_root = custom_bool(prim, &prop("includeRoot")); - let expression = prim - .attribute(prop("membershipExpression")) - .get_at::(eval_time()) - .ok() - .flatten() - .is_some(); + let Opinions { + includes, + excludes, + include_root, + expression, + .. + } = opinions(prim, name); if expression { warn!( "{}: collection:{name} authors membershipExpression, which crust does not \ @@ -153,13 +145,8 @@ fn opinions(prim: &Prim, name: &str) -> Opinions { includes: targets("includes"), excludes: targets("excludes"), include_root: custom_bool(prim, &prop("includeRoot")), - expansion_rule: super::attrs::custom_token(prim, &prop("expansionRule")), - expression: prim - .attribute(prop("membershipExpression")) - .get_at::(eval_time()) - .ok() - .flatten() - .is_some(), + expansion_rule: custom_token(prim, &prop("expansionRule")), + expression: prim_value(prim, &prop("membershipExpression")).is_some(), } } diff --git a/crates/crust-core/src/scene/usd_import/lights.rs b/crates/crust-core/src/scene/usd_import/lights.rs index b8d28566..152dbb41 100644 --- a/crates/crust-core/src/scene/usd_import/lights.rs +++ b/crates/crust-core/src/scene/usd_import/lights.rs @@ -1,12 +1,9 @@ //! UsdLux lights → [`LightList`] entries (and their emissive scene geometry). -use std::path::Path; use std::sync::Arc; -use std::time::Instant; use crust_rt::Geometry; use glam::{Affine3A, Mat3A, Mat4 as GMat4, Vec3, Vec3A}; -use openusd::sdf; use openusd::usd::{Prim, Stage}; use openusd_schemas::lux::{ CylinderLight, DiskLight, DistantLight as UsdDistantLight, DomeLight, Light as UsdLight, @@ -23,12 +20,11 @@ use crate::lux::{IesShaping, Shaping, distant_illuminance, distant_size_factor}; use crate::material::Emissive; use crate::rt_world::WorldBuilder; +use super::assets::{asset_path, cached_asset, timed_asset}; use super::attrs::{ - attr_bool, attr_color3f, attr_f32, attr_own_color_space, custom_bool, custom_color, custom_f32, - custom_token, in_working, infinite_light_escape_mask, light_ray_mask, + attr_bool, attr_f32, attr_own_color_space, attr_vec3, custom_bool, custom_color, custom_f32, + custom_token, decode_text, in_working, infinite_light_escape_mask, light_ray_mask, value_at, }; -use super::materials::asset_value_path; -use super::time::eval_time; use super::{ImportCaches, ImportCtx}; /// The `LightAPI` quantities every UsdLux light shares. @@ -60,17 +56,18 @@ fn lux_params(prim: &Prim, light: &impl UsdLight, working: Space) -> LuxParams { }; let intensity = finite("intensity", attr_f32(&light.intensity_attr()), 1.0); let exposure = finite("exposure", attr_f32(&light.exposure_attr()), 0.0); - let color = match attr_color3f(&light.color_attr()) { - Some(c) if c.iter().any(|x| !x.is_finite()) => { + let color = match attr_vec3(&light.color_attr()) { + Some(c) if !c.is_finite() => { warn!( - "{}: inputs:color = {c:?} is not finite — using its fallback (1, 1, 1)", - prim.path() + "{}: inputs:color = {:?} is not finite — using its fallback (1, 1, 1)", + prim.path(), + c.to_array() ); - [1.0; 3] + Vec3A::ONE } - c => c.unwrap_or([1.0; 3]), + c => c.unwrap_or(Vec3A::ONE), }; - let color = in_working(&light.color_attr(), Vec3A::from_array(color), working); + let color = in_working(&light.color_attr(), color, working); let gain = intensity * 2f32.powf(exposure); let mut emission = color * gain; @@ -173,31 +170,23 @@ fn lux_shaping( ) .unwrap_or(0.0); - let ies_file = prim - .attribute("inputs:shaping:ies:file") - .get_at::(eval_time()) - .ok() - .flatten() - .and_then(|v| asset_value_path(&v, caches.stage_path)); + let ies_file = asset_path( + &prim.attribute("inputs:shaping:ies:file"), + caches.stage_path, + ); if let Some(path) = ies_file { - let profile = match caches.ies.get(&path) { - Some(cached) => cached.clone(), - None => { - let started = Instant::now(); - let loaded = caches.assets.load_ies(&path); - caches.asset_time += started.elapsed(); - if loaded.is_none() { - warn!( - "{}: could not load IES profile {} — the light renders \ - without it", - prim.path(), - path.display() - ); - } - caches.ies.insert(path, loaded.clone()); - loaded + let assets = caches.assets; + let profile = cached_asset(&mut caches.ies, &mut caches.asset_time, path, |path| { + let loaded = assets.load_ies(path); + if loaded.is_none() { + warn!( + "{}: could not load IES profile {} — the light renders without it", + prim.path(), + path.display() + ); } - }; + loaded + }); shaping.ies = profile.map(|profile| IesShaping { profile, angle_scale: finite( @@ -485,38 +474,34 @@ fn texture_color_space(file: &openusd::usd::Attribute, working: Space) -> crate: /// `RectLight`'s `inputs:texture:file`, decoded by the host. Cached by /// resolved path: a rig commonly reuses one card texture on many lights. fn rect_light_texture(prim: &Prim, caches: &mut ImportCaches) -> Option> { - let value = prim - .attribute("inputs:texture:file") - .get_at::(eval_time()) - .ok() - .flatten()?; - let path = asset_value_path(&value, caches.stage_path)?; - let space = texture_color_space(&prim.attribute("inputs:texture:file"), caches.working); - let key = (path, space); - if let Some(cached) = caches.light_textures.get(&key) { - return cached.clone(); - } - let path = &key.0; - let started = Instant::now(); - let loaded = caches.assets.load_light_texture(path, space); - caches.asset_time += started.elapsed(); - match &loaded { - Some(t) => debug!( - "RectLight {}: texture {} ({}x{})", - prim.path(), - path.display(), - t.width(), - t.height() - ), - None => warn!( - "RectLight at {}: could not load inputs:texture:file {} — the light \ - emits its uniform colour", - prim.path(), - path.display() - ), - } - caches.light_textures.insert(key, loaded.clone()); - loaded + let file = prim.attribute("inputs:texture:file"); + let path = asset_path(&file, caches.stage_path)?; + let space = texture_color_space(&file, caches.working); + let assets = caches.assets; + cached_asset( + &mut caches.light_textures, + &mut caches.asset_time, + (path, space), + |(path, space)| { + let loaded = assets.load_light_texture(path, *space); + match &loaded { + Some(t) => debug!( + "RectLight {}: texture {} ({}x{})", + prim.path(), + path.display(), + t.width(), + t.height() + ), + None => warn!( + "RectLight at {}: could not load inputs:texture:file {} — the light \ + emits its uniform colour", + prim.path(), + path.display() + ), + } + loaded + }, + ) } /// `UsdLuxRectLight`: a `width × height` rectangle (1 × 1) in the local XY @@ -717,24 +702,17 @@ pub(super) fn emit_dome_light( // `normalize` does not apply to a dome (its sizeFactor is 1). let tint = lux_params(prim, light, working).emission; - let format = light - .texture_format_attr() - .get_at::(eval_time()) - .ok() - .flatten() - .and_then(|v| match v { - sdf::Value::Token(t) => Some(t.to_string()), - _ => None, - }); - let map = match dome_texture_path(light, stage_path) { + let format = value_at(&light.texture_format_attr()).and_then(decode_text); + // The authoring layer anchors the path, not the root layer: the Moana + // island's lights author `../textures/islandsun.exr` relative to + // `usd/island.usda` (see `asset_path`). + let map = match asset_path(&light.texture_file_attr(), stage_path) { Some(texture) => match format.as_deref() { // `automatic` infers from the image; for the equirectangular // images a dome light normally carries that means latlong. None | Some("latlong") | Some("automatic") => { - let started = Instant::now(); let space = texture_color_space(&light.texture_file_attr(), working); - let loaded = assets.load_environment(&texture, space); - *asset_time += started.elapsed(); + let loaded = timed_asset(asset_time, || assets.load_environment(&texture, space)); if loaded.is_none() { warn!( "DomeLight at {}: could not load {} — falling back to \ @@ -792,20 +770,3 @@ fn tag_last(lights: &mut LightList, prim: &Prim) { lights.set_lpe_tag(index, Some(&tag)); } } - -/// The dome's `inputs:texture:file` as a filesystem path. -/// -/// Goes through [`asset_value_path`], which prefers openusd's `resolved_path()` -/// — anchored against the layer that *authored* the path, not the root layer. -/// That distinction only shows up once a stage has depth: the Moana island's -/// lights author `../textures/islandsun.exr` relative to `usd/island.usda`, so -/// a root layer sitting anywhere else would otherwise resolve it against the -/// wrong directory and silently fall back to the dome's uniform colour. -fn dome_texture_path(light: &DomeLight, stage_path: &Path) -> Option { - let value = light - .texture_file_attr() - .get_at::(eval_time()) - .ok() - .flatten()?; - asset_value_path(&value, stage_path) -} diff --git a/crates/crust-core/src/scene/usd_import/materials.rs b/crates/crust-core/src/scene/usd_import/materials.rs index 488646cf..95f3f91a 100644 --- a/crates/crust-core/src/scene/usd_import/materials.rs +++ b/crates/crust-core/src/scene/usd_import/materials.rs @@ -1,9 +1,8 @@ //! Material binding and dispatch: `MaterialBindingAPI` resolution, the //! per-stage material cache, and the decoders for `crust:openpbr`, -//! `PxrDisneyBsdf`, MaterialX references and asset paths. +//! `PxrDisneyBsdf` and MaterialX references. use std::collections::HashMap; -use std::path::Path; use std::sync::Arc; use std::time::Instant; @@ -19,9 +18,12 @@ use tracing::{debug, warn}; use crate::color::Space; use crate::material::{DispRemap, Displacement, DisplacementValue, Material, OpenPBR}; -use super::attrs::{attr_own_color_space, custom_f32, in_working}; +use super::assets::{asset_path, load_ptex, load_uv_texture}; +use super::attrs::{ + attr_bool, attr_f32, attr_own_color_space, attr_vec3, custom_color3, custom_f32, custom_token, + decode_number, decode_text, in_working, value_at, +}; use super::preview::{preview_displacement, preview_surface_material}; -use super::time::eval_time; use super::{ImportCaches, prim_at}; /// Memoizes resolved materials by binding path (and shares one default), @@ -264,13 +266,9 @@ fn resolve_displacement( fn child_shader(stage: &Stage, mat_path: &sdf::Path, id: &str) -> Option { let children = prim_at(stage, mat_path.clone()).children().ok()?; // A token or a string, as `has_shader_id` accepts. - let child = children.iter().find(|c| { - match c.attribute("info:id").get_at::(eval_time()) { - Ok(Some(sdf::Value::Token(t))) => t.as_str() == id, - Ok(Some(sdf::Value::String(t))) => t == id, - _ => false, - } - })?; + let child = children + .iter() + .find(|c| custom_token(c, "info:id").as_deref() == Some(id))?; Shader::get(stage, child.path().clone()).ok().flatten() } @@ -282,21 +280,11 @@ fn input_value(shader: &Shader, name: &str) -> Option { .value_producing_attributes(ProducerFilter::Any) .ok()? .into_iter() - .find_map(|a| { - a.attribute() - .get_at::(eval_time()) - .ok() - .flatten() - }) + .find_map(|a| value_at(a.attribute())) } fn input_f32(shader: &Shader, name: &str) -> Option { - match input_value(shader, name)? { - sdf::Value::Float(f) => Some(f), - sdf::Value::Double(d) => Some(d as f32), - sdf::Value::Int(i) => Some(i as f32), - _ => None, - } + input_value(shader, name).and_then(decode_number) } /// The shader whose output drives `shader.inputs:`, if one does. @@ -410,7 +398,7 @@ fn pxr_displacement( .value_producing_attributes(ProducerFilter::Any) .ok()? .into_iter() - .find_map(|a| attribute_asset_path(a.attribute(), caches.stage_path))?; + .find_map(|a| asset_path(a.attribute(), caches.stage_path))?; maps.push(crate::PtexRef(load_ptex( &file, crate::ColorSpace::RAW, @@ -574,18 +562,7 @@ pub(super) fn shader_info_id(shader: &Shader) -> Option { } // Fallback for older openusd revisions or shaders that author info:id // via a raw attribute rather than the schema helper. - shader - .attribute("info:id") - .get_at::(eval_time()) - .ok() - .flatten() - .and_then(|v| match v { - // `Token` carries an interned `tf::Token`, `String` a plain - // `String`, so the two arms cannot bind the same name. - sdf::Value::Token(t) => Some(t.as_str().to_owned()), - sdf::Value::String(t) => Some(t), - _ => None, - }) + value_at(&shader.attribute("info:id")).and_then(decode_text) } /// Whether the material has a child `Shader` prim with this `info:id`. @@ -598,13 +575,9 @@ fn has_shader_id(stage: &Stage, mat_path: &sdf::Path, id: &str) -> bool { let Ok(children) = prim_at(stage, mat_path.clone()).children() else { return false; }; - children.iter().any( - |c| match c.attribute("info:id").get_at::(eval_time()) { - Ok(Some(sdf::Value::Token(t))) => t.as_str() == id, - Ok(Some(sdf::Value::String(t))) => t == id, - _ => false, - }, - ) + children + .iter() + .any(|c| custom_token(c, "info:id").as_deref() == Some(id)) } /// Maps RenderMan's `PxrDisneyBsdf` onto [`OpenPBR`]. @@ -637,7 +610,7 @@ fn disney_to_openpbr( // Called with the whole attribute name, `inputs:` included: a literal, so // reading an input allocates no name. let f = |n: &str| custom_f32(&prim, n); - let c = |n: &str| custom_vec3(&prim, n); + let c = |n: &str| custom_color3(&prim, n); let mut o = OpenPBR { luma: caches.luma, @@ -758,7 +731,7 @@ fn mtlx_reference(stage: &Stage, mat_path: &sdf::Path) -> Option<(std::path::Pat continue; } // Anchored against the *authoring layer's* directory, the same - // rule `asset_value_path` follows for textures and for the + // rule `asset_path` follows for textures and for the // same reason: `Looks/teapot_ceramic_ldX.mtlx` is relative to // `teapot.usda`, which need not be the stage root. let base = std::path::Path::new(&id).parent()?.to_path_buf(); @@ -864,34 +837,6 @@ fn load_mtlx_material( } } -/// Opens a UV texture through the host, memoized by resolved path and colour -/// space. The caller maps its own vocabulary onto the space — MaterialX's -/// `colorspace` through [`crate::ColorSpace::from_mtlx`], UsdUVTexture's -/// `sourceColorSpace` through [`crate::ColorSpace::from_usd`] — since the two -/// disagree on what an absent attribute means. -pub(super) fn load_uv_texture( - path: &std::path::Path, - space: crate::ColorSpace, - caches: &mut ImportCaches<'_>, -) -> Option> { - let key = (path.to_string_lossy().into_owned(), space); - if let Some(hit) = caches.materials.textures.get(&key) { - return hit.clone(); - } - let started = Instant::now(); - let loaded = caches.assets.load_texture(path, space); - let elapsed = started.elapsed(); - caches.asset_time += elapsed; - if loaded.is_none() { - debug!( - "Texture {} ({space:?}) not loadable by the host", - path.display() - ); - } - caches.materials.textures.insert(key, loaded.clone()); - loaded -} - /// The per-face colour texture a material binds, if any. /// /// `inputs:surfaceMap` is the interface input both of the island's Ptex shader @@ -904,124 +849,12 @@ pub(super) fn material_ptex( caches: &mut ImportCaches<'_>, ) -> Option { let prim = prim_at(stage, mat_path.clone()); - let value = prim - .attribute("inputs:surfaceMap") - .get_at::(eval_time()) - .ok() - .flatten()?; - let path = asset_value_path(&value, caches.stage_path)?; + let path = asset_path(&prim.attribute("inputs:surfaceMap"), caches.stage_path)?; let space = crate::ColorSpace::new(Space::G22_REC709, caches.working); load_ptex(&path, space, caches).map(crate::PtexRef) } -/// Opens a Ptex file through the host, once per `(resolved path, space)`. -/// -/// Keyed on the resolved filesystem path, which — unlike a prototype-scoped -/// scene path — is stable across the streaming importer's stages, so one -/// texture is opened once however many materials or chunks reference it. -/// The space is in the key because the decode happens at open: a file read -/// both as colour and as displacement is two textures, which is correct and -/// rare. Negative results are cached too: a 600 MB file that failed to open -/// should not be retried per material. -pub(super) fn load_ptex( - path: &std::path::Path, - space: crate::ColorSpace, - caches: &mut ImportCaches<'_>, -) -> Option> { - let key = (path.to_string_lossy().into_owned(), space); - if let Some(hit) = caches.materials.ptex.get(&key) { - return hit.clone(); - } - let started = Instant::now(); - let loaded = caches.assets.load_ptex(path, space); - caches.asset_time += started.elapsed(); - caches.materials.ptex.insert(key, loaded.clone()); - loaded -} - -/// An `asset`-valued attribute as a filesystem path. -/// -/// openusd anchors default-sourced asset paths against the layer that authored -/// them and reports the result in `resolved_path` — which is what makes a -/// production stage's `../../../textures/foo.ptx` work at all, since the layer -/// authoring it is nested several directories below the root. The authored -/// string is only a fallback, anchored against the root layer. -pub(super) fn asset_value_path( - value: &sdf::Value, - stage_path: &Path, -) -> Option { - let (authored, resolved) = match value { - sdf::Value::AssetPath(p) => (p.as_str().to_string(), p.resolved_path()), - sdf::Value::String(p) => (p.clone(), None), - _ => return None, - }; - if let Some(r) = resolved - && !r.is_empty() - { - return Some(std::path::PathBuf::from(r)); - } - if authored.is_empty() { - return None; - } - let candidate = std::path::Path::new(&authored); - if candidate.is_absolute() { - return Some(candidate.to_path_buf()); - } - Some( - stage_path - .parent() - .unwrap_or_else(|| std::path::Path::new(".")) - .join(candidate), - ) -} - -/// [`asset_value_path`] for an attribute, anchoring an **unresolved** relative -/// path against the layer that authored it rather than against the root layer. -/// -/// openusd anchors every asset value against its authoring layer, but reports -/// the anchored path only when it names a file that exists — and a -/// ``-tokened texture path never does, since it names a set. So such a -/// value arrives with no `resolved_path` at all, and anchoring it against the -/// root layer was wrong for any texture authored in a sublayer or reference: -/// ALab's look layers sit five directories below `entry.usda` and author -/// `@../../texture/….exr@`, and 1 718 texture sets failed to load. The -/// strongest spec in the attribute's property stack is the layer whose opinion -/// supplies the value, which is exactly what USD anchors against. -pub(super) fn attribute_asset_path( - attr: &openusd::usd::Attribute, - stage_path: &Path, -) -> Option { - let value = attr.get_at::(eval_time()).ok().flatten()?; - let resolved = matches!(&value, sdf::Value::AssetPath(p) - if p.resolved_path().is_some_and(|r| !r.is_empty())); - let authored = value.as_str().map(str::to_owned); - if !resolved - && let Some(authored) = authored.filter(|a| !a.is_empty() && Path::new(a).is_relative()) - && let Some(layer_dir) = attr - .property_stack() - .ok() - .and_then(|stack| stack.into_iter().next()) - .and_then(|site| Path::new(&site.layer).parent().map(Path::to_path_buf)) - .filter(|d| !d.as_os_str().is_empty()) - { - return Some(layer_dir.join(authored)); - } - asset_value_path(&value, stage_path) -} - -fn custom_vec3(prim: &Prim, name: &str) -> Option { - let v = prim - .attribute(name) - .get_at::(eval_time()) - .ok()??; - match v { - sdf::Value::Vec3f(p) => Some(Vec3A::new(p.x, p.y, p.z)), - sdf::Value::Vec3d(p) => Some(Vec3A::new(p.x as f32, p.y as f32, p.z as f32)), - _ => None, - } -} - /// Decode a `crust:openpbr` shader into the OpenPBR material. Every input /// name is camelCase mirror of the Rust snake_case, e.g. `base_color` → /// `inputs:baseColor`, `subsurface_radius_scale` → `inputs:subsurfaceRadiusScale`. @@ -1031,15 +864,17 @@ fn decode_crust_openpbr(shader: &Shader, working: Space, luma: utils::Luma) -> A ..OpenPBR::default() }; - let f = |n: &str, d: f32| shader_input_f32(shader, n).unwrap_or(d); + // Inputs by their whole attribute name (`inputs:roughness`): the names are + // literals, so none is built per read. + let f = |n: &str, d: f32| attr_f32(&shader.attribute(n)).unwrap_or(d); // Every colour is authored in the working space unless its `colorSpace` // metadatum names another (`in_working`); `v` reads a vector that is not // a colour and is never converted. let c = |n: &str, d: Vec3A| { - shader_input_vec3(shader, n).map_or(d, |v| in_working(&shader.attribute(n), v, working)) + attr_vec3(&shader.attribute(n)).map_or(d, |v| in_working(&shader.attribute(n), v, working)) }; - let v = |n: &str, d: Vec3A| shader_input_vec3(shader, n).unwrap_or(d); - let b = |n: &str, d: bool| shader_input_bool(shader, n).unwrap_or(d); + let v = |n: &str, d: Vec3A| attr_vec3(&shader.attribute(n)).unwrap_or(d); + let b = |n: &str, d: bool| attr_bool(&shader.attribute(n)).unwrap_or(d); // Base o.base_weight = f("inputs:baseWeight", o.base_weight); @@ -1116,40 +951,3 @@ fn decode_crust_openpbr(shader: &Shader, working: Space, luma: utils::Luma) -> A Arc::new(o) } - -/// A shader input by its whole attribute name (`inputs:roughness`) — the -/// callers pass literals, so no name is built per read. -fn shader_input_f32(shader: &Shader, attr_name: &str) -> Option { - let v = shader - .attribute(attr_name) - .get_at::(eval_time()) - .ok()??; - match v { - sdf::Value::Float(f) => Some(f), - sdf::Value::Double(d) => Some(d as f32), - _ => None, - } -} - -fn shader_input_bool(shader: &Shader, attr_name: &str) -> Option { - let v = shader - .attribute(attr_name) - .get_at::(eval_time()) - .ok()??; - match v { - sdf::Value::Bool(b) => Some(b), - _ => None, - } -} - -fn shader_input_vec3(shader: &Shader, attr_name: &str) -> Option { - let v = shader - .attribute(attr_name) - .get_at::(eval_time()) - .ok()??; - match v { - sdf::Value::Vec3f(p) => Some(Vec3A::new(p.x, p.y, p.z)), - // USD encodes color3f as an sdf::Value::Vec3f — no dedicated variant. - _ => None, - } -} diff --git a/crates/crust-core/src/scene/usd_import/mesh.rs b/crates/crust-core/src/scene/usd_import/mesh.rs index e3e578fa..cba3a5f3 100644 --- a/crates/crust-core/src/scene/usd_import/mesh.rs +++ b/crates/crust-core/src/scene/usd_import/mesh.rs @@ -24,7 +24,10 @@ use crate::scene::displace::{self, VertexChart}; use crate::scene::subdiv; use super::adaptive::{self, Aabb, Cull, ScreenRate}; -use super::attrs::{custom_f32, custom_i32, prim_motion_translate, prim_ray_mask}; +use super::attrs::{ + custom_f32, custom_i32, custom_i32_array, decode_f32_array, decode_i32_array, + decode_vec3f_array, prim_motion_translate, prim_ray_mask, prim_value, value_at, +}; use super::materials::BoundMaterial; use super::time::eval_time; @@ -847,31 +850,9 @@ fn bake_indices(mut tris: Vec<[u32; 3]>, l2w: &Affine3A) -> Vec<[u32; 3]> { /// Reads a mesh prim's authored arrays. `None` when any of the three /// required attributes is missing. pub(super) fn mesh_arrays(mesh: &UsdMesh) -> Option<(Vec, Vec, Vec)> { - let int_vec = |v: sdf::Value| match v { - sdf::Value::IntVec(v) => Some(v), - _ => None, - }; - let points = match mesh - .points_attr() - .get_at::(eval_time()) - .ok() - .flatten()? - { - sdf::Value::Vec3fVec(v) => v, - _ => return None, - }; - let counts = int_vec( - mesh.face_vertex_counts_attr() - .get_at::(eval_time()) - .ok() - .flatten()?, - )?; - let indices = int_vec( - mesh.face_vertex_indices_attr() - .get_at::(eval_time()) - .ok() - .flatten()?, - )?; + let points = value_at(&mesh.points_attr()).and_then(decode_vec3f_array)?; + let counts = value_at(&mesh.face_vertex_counts_attr()).and_then(decode_i32_array)?; + let indices = value_at(&mesh.face_vertex_indices_attr()).and_then(decode_i32_array)?; Some((points, counts, indices)) } @@ -924,12 +905,7 @@ pub(super) fn mesh_uvs(prim: &Prim, preferred: Option<&str>) -> Option "primvars:st0", "primvars:UVMap", ]) { - let value = prim - .attribute(name) - .get_at::(eval_time()) - .ok() - .flatten(); - let values = match value { + let values = match prim_value(prim, name) { // `texCoord2f[]` and `float2[]` are the same bits; which one an // exporter writes is a matter of taste. Some(sdf::Value::Vec2fVec(v)) => v.iter().map(|p| [p.x, p.y]).collect::>(), @@ -938,15 +914,7 @@ pub(super) fn mesh_uvs(prim: &Prim, preferred: Option<&str>) -> Option if values.is_empty() { continue; } - let indices = match prim - .attribute(format!("{name}:indices")) - .get_at::(eval_time()) - .ok() - .flatten() - { - Some(sdf::Value::IntVec(v)) => Some(v), - _ => None, - }; + let indices = custom_i32_array(prim, &format!("{name}:indices")); // USD's fallback interpolation for a primvar is `constant`, but for // `st` in practice it is always authored; treating an unauthored // metadatum as faceVarying would mis-index a vertex-interpolated @@ -1168,18 +1136,16 @@ pub(super) fn mesh_source( } }; - let int_array = - |attr: openusd::usd::Attribute| match attr.get_at::(eval_time()).ok().flatten() - { - Some(sdf::Value::IntVec(v)) => v, - _ => Vec::new(), - }; - let float_array = - |attr: openusd::usd::Attribute| match attr.get_at::(eval_time()).ok().flatten() - { - Some(sdf::Value::FloatVec(v)) => v, - _ => Vec::new(), - }; + let int_array = |attr| { + value_at(&attr) + .and_then(decode_i32_array) + .unwrap_or_default() + }; + let float_array = |attr| { + value_at(&attr) + .and_then(decode_f32_array) + .unwrap_or_default() + }; let crease_indices = int_array(mesh.crease_indices_attr()); let crease_lengths = int_array(mesh.crease_lengths_attr()); let crease_sharpnesses = float_array(mesh.crease_sharpnesses_attr()); diff --git a/crates/crust-core/src/scene/usd_import/mod.rs b/crates/crust-core/src/scene/usd_import/mod.rs index bbf16408..11c3d206 100644 --- a/crates/crust-core/src/scene/usd_import/mod.rs +++ b/crates/crust-core/src/scene/usd_import/mod.rs @@ -58,6 +58,7 @@ use openusd_schemas::lux::{ }; mod adaptive; +mod assets; mod attrs; mod camera; mod instancing; @@ -75,7 +76,8 @@ mod xform; use adaptive::{Frustum, ScreenRate}; use attrs::{ - custom_token, resolve_adaptive_max_level, resolve_subdiv_edge_length, resolve_subdiv_level, + custom_token, prim_value, resolve_adaptive_max_level, resolve_subdiv_edge_length, + resolve_subdiv_level, }; use camera::{build_camera, screen_projection}; use instancing::{ProtoPart, emit_native_instance, emit_point_instancer}; @@ -93,9 +95,9 @@ use settings::{ render_settings_subdiv_level, }; use shapes::{emit_curves, emit_sphere}; -use time::{EvalTimeScope, eval_time}; +use time::EvalTimeScope; use volume::emit_volume; -use xform::{local_matrix_at, resets_xform_stack_at}; +use xform::compose_with_parent; /// `Stage::prim` for a path that is already an `sdf::Path`. /// @@ -174,26 +176,17 @@ fn subtree_roots(stage: &Stage) -> Vec { } /// Counts the native placements of every prototype on `stage`, per top-level -/// subtree, into `caches.placements`: the same walk and pruning as -/// [`traverse_into`] (abstract, inactive, non-render purpose, invisible), not +/// subtree, into `caches.placements`: the same walk and pruning +/// ([`prune_reason`]) as [`traverse_into`], not /// descending into an instance or a `PointInstancer`, whose contents the /// traversal does not reach directly either. fn count_placements(stage: &Stage, caches: &mut ImportCaches<'_>) { let mut stack = vec![(prim_at(stage, sdf::Path::abs_root()), GMat4::IDENTITY)]; while let Some((prim, parent_world)) = stack.pop() { - if prim.is_abstract().unwrap_or(false) - || !prim.is_active().unwrap_or(true) - || non_render_purpose(&prim).is_some() - || is_invisible(&prim) - { + if prune_reason(&prim, WalkScope::Stage).is_some() { continue; } - let local = local_matrix_at(stage, &prim); - let world = if resets_xform_stack_at(stage, &prim) { - local - } else { - parent_world * local - }; + let world = compose_with_parent(stage, &prim, parent_world); if prim.is_instance().unwrap_or(false) && let Ok(Some(proto)) = prim.prototype() { @@ -269,30 +262,13 @@ fn traverse_into(stage: &Stage, root: Prim, root_xf: GMat4, ctx: &mut ImportCtx) let mut stack: Vec<(Prim, GMat4, bool)> = vec![(root, root_xf, false)]; while let Some((prim, parent_world, parent_hidden)) = stack.pop() { - // `class` prims (and their descendants) describe geometry that - // exists only to be referenced or instanced — they are never - // rendered in their own right. Prototypes reach the same prims - // through `collect_proto_parts`, which deliberately does not apply - // this rule. - if prim.is_abstract().unwrap_or(false) { - debug!("Skipping abstract (class) prim {}", prim.path()); - continue; - } - // USD prunes an inactive prim and its whole namespace subtree from - // the composed scene — the standard way a stage disables geometry - // (e.g. an LOD or a too-dense archive) without editing its source. - if !prim.is_active().unwrap_or(true) { - debug!("Skipping inactive prim {}", prim.path()); - continue; - } - if let Some(purpose) = non_render_purpose(&prim) { - debug!("Skipping {purpose}-purpose prim {}", prim.path()); + let pruned = prune_reason(&prim, WalkScope::Stage); + if let Some(reason) = pruned.filter(|&r| r != Prune::Invisible) { + debug!("Skipping {reason} prim {}", prim.path()); continue; } - let local = local_matrix_at(stage, &prim); - let resets = resets_xform_stack_at(stage, &prim); - let this_world = if resets { local } else { parent_world * local }; + let this_world = compose_with_parent(stage, &prim, parent_world); // An invisible subtree draws nothing and lights nothing, but is still // walked for cameras: a camera's own visibility only hides its gizmo @@ -300,7 +276,7 @@ fn traverse_into(stage: &Stage, root: Prim, root_xf: GMat4, ctx: &mut ImportCtx) // Pruning it outright would make `--camera` fail on it and move the // first-camera fallback. Instances are not entered — crust never // takes a camera from a prototype. - let hidden = parent_hidden || is_invisible(&prim); + let hidden = parent_hidden || pruned == Some(Prune::Invisible); if hidden { if !parent_hidden { debug!("Skipping invisible prim {} and its subtree", prim.path()); @@ -463,22 +439,87 @@ fn visit_camera(stage: &Stage, prim: &Prim, ctx: &mut ImportCtx) -> bool { true } -/// Whether `prim` authors `visibility = "invisible"` at the evaluated time. +/// Why a traversal leaves a prim and its whole subtree out of the render — +/// see [`prune_reason`]. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub(super) enum Prune { + /// A `class` prim: it exists only to be referenced or instanced. + Abstract, + /// `active = false`. + Inactive, + /// A purpose a final render does not draw (`"proxy"` / `"guide"`). + Purpose(&'static str), + /// `visibility = "invisible"`. + Invisible, +} + +impl std::fmt::Display for Prune { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Prune::Abstract => f.write_str("abstract (class)"), + Prune::Inactive => f.write_str("inactive"), + Prune::Purpose(p) => write!(f, "{p}-purpose"), + Prune::Invisible => f.write_str("invisible"), + } + } +} + +/// Where a prim is met, which decides whether [`Prune::Abstract`] applies. +#[derive(Clone, Copy, PartialEq, Eq)] +pub(super) enum WalkScope { + /// The stage's own namespace: the top-level traversal and the placement + /// count that must agree with it. + Stage, + /// Inside a prototype, reached through an instance. A prototype is + /// commonly authored under a `class` prim, so abstractness does not prune + /// here. + Prototype, +} + +/// The one pruning rule every walk applies, so that the traversal, the +/// placement count and the prototype walk cannot disagree about which prims +/// exist. /// -/// Visibility is inherited and cannot be undone below: an `invisible` -/// ancestor hides its whole subtree whatever the descendants author (their -/// only other value, `inherited`, defers to it), so as with `purpose` the -/// subtree is decided where the opinion is authored (UsdGeomImageable's -/// `ComputeVisibility`). It applies to lights as it does to geometry: an -/// invisible light does not illuminate. ALab's rig parks three interior -/// fills/bounces and a debug dome this way, and all four used to light the -/// shot. -fn is_invisible(prim: &Prim) -> bool { - prim.attribute("visibility") - .get_at::(eval_time()) - .ok() - .flatten() +/// Each rule prunes a whole subtree, decided where the opinion is authored: +/// +/// - **abstract** (stage scope only) — `class` prims describe geometry that +/// is never rendered in its own right. +/// - **inactive** — USD prunes an inactive prim and its namespace subtree +/// from the composed scene, the standard way a stage disables geometry (an +/// LOD, a too-dense archive) without editing its source. +/// - **non-render purpose** — a render draws `default` and `render` purpose +/// only (UsdGeomImageable). Purpose is inherited and a non-default one on an +/// ancestor wins, so pruning where it is authored is exactly +/// `ComputePurpose` for a walk from the root. Without it a production asset +/// renders twice: ALab publishes every asset with a `GEO_PROXY` scope +/// (`purpose = "proxy"`) next to its `GEO`, and the proxies drew as grey +/// duplicates of 1 505 meshes. +/// - **invisible** — visibility is inherited and cannot be undone below (the +/// only other value, `inherited`, defers to it), as `ComputeVisibility` +/// has it. It applies to lights as to geometry: ALab's rig parks three +/// interior fills/bounces and a debug dome this way, and all four used to +/// light the shot. The top-level traversal still walks an invisible +/// subtree for cameras (see [`traverse_into`]). +/// +/// Checked in that order, so the reason reported is the first that applies. +pub(super) fn prune_reason(prim: &Prim, scope: WalkScope) -> Option { + if scope == WalkScope::Stage && prim.is_abstract().unwrap_or(false) { + return Some(Prune::Abstract); + } + if !prim.is_active().unwrap_or(true) { + return Some(Prune::Inactive); + } + let token = |name: &str| prim_value(prim, name); + if let Some(v) = token("purpose") { + match v.as_str() { + Some("proxy") => return Some(Prune::Purpose("proxy")), + Some("guide") => return Some(Prune::Purpose("guide")), + _ => {} + } + } + token("visibility") .is_some_and(|v| v.as_str() == Some("invisible")) + .then_some(Prune::Invisible) } /// Drops a stage the traversal is done with, or — for a single-stage import @@ -810,8 +851,10 @@ pub(crate) fn load_scene( ); } if let Some(rate) = ctx.caches.meshes.subdiv.adaptive { - // The two reads of one camera must agree, or every level was chosen - // for a viewpoint the render does not use. + // Both reads derive from one `CameraFrame`, but adaptive subdivision + // may read it off a different stage (the unloaded index while + // streaming); the two must still agree, or every level was chosen for + // a viewpoint the render does not use. debug_assert!( (Vec3::from(camera.origin()) - rate.eye).length() <= 1e-4 * rate.eye.length().max(1.0), "adaptive subdivision read the camera at {:?}, the render camera is at {:?}", @@ -1088,27 +1131,3 @@ impl<'a> ImportCaches<'a> { } } } - -/// `Some("proxy")` / `Some("guide")` when `prim` authors a purpose a final -/// render does not draw. -/// -/// A render draws `default` and `render` purpose only (UsdGeomImageable). -/// Purpose is inherited, and a non-default purpose on an ancestor wins over -/// whatever its descendants author, so pruning the subtree where the purpose -/// is authored is exactly `ComputePurpose` for a traversal that descends from -/// the root — the same shape as the `active = false` pruning beside it. -/// Without it a production asset renders twice: ALab publishes every asset -/// with a `GEO_PROXY` scope (`purpose = "proxy"`, bound only for `preview`) -/// next to its `GEO`, and the proxies drew as grey duplicates of 1 505 meshes. -fn non_render_purpose(prim: &Prim) -> Option<&'static str> { - let value = prim - .attribute("purpose") - .get_at::(eval_time()) - .ok() - .flatten()?; - match value.as_str()? { - "proxy" => Some("proxy"), - "guide" => Some("guide"), - _ => None, - } -} diff --git a/crates/crust-core/src/scene/usd_import/preview.rs b/crates/crust-core/src/scene/usd_import/preview.rs index d49e0b0c..59e172e9 100644 --- a/crates/crust-core/src/scene/usd_import/preview.rs +++ b/crates/crust-core/src/scene/usd_import/preview.rs @@ -15,9 +15,9 @@ use crate::color::Space; use crate::material::{Displacement, DisplacementValue, Material, OpenPBR}; use super::ImportCaches; -use super::attrs::{attr_own_color_space, in_working}; -use super::materials::{attribute_asset_path, load_uv_texture, material_ptex, shader_info_id}; -use super::time::eval_time; +use super::assets::{asset_path, load_uv_texture}; +use super::attrs::{attr_own_color_space, decode_float4, in_working, value_at}; +use super::materials::{material_ptex, shader_info_id}; /// A `UsdPreviewSurface` material: its constants as an [`OpenPBR`], wrapped in /// a [`crate::PreviewSurface`] when any input is driven by a `UsdUVTexture`. @@ -222,22 +222,17 @@ fn preview_uv_input( // The value an input carries, connection followed. let value = |input: &shade::Input| -> Option { let produced = input.value_producing_attributes(ProducerFilter::Any).ok()?; - produced - .first()? - .attribute() - .get_at::(eval_time()) - .ok() - .flatten() + value_at(produced.first()?.attribute()) }; let token = |input: &str| value(&tex.input(input)).and_then(|v| v.as_str().map(str::to_owned)); - let float4 = |input: &str| value(&tex.input(input)).and_then(|v| sdf_float4(&v)); + let float4 = |input: &str| value(&tex.input(input)).and_then(decode_float4); let file = tex .input(tk::TEX_FILE) .value_producing_attributes(ProducerFilter::Any) .ok() .and_then(|p| p.into_iter().next()) - .and_then(|a| attribute_asset_path(a.attribute(), caches.stage_path)); + .and_then(|a| asset_path(a.attribute(), caches.stage_path)); let Some(file) = file else { warn!( "UsdUVTexture {}: no inputs:file — {name} keeps its constant", @@ -312,14 +307,7 @@ fn preview_uv_input( // roughness map the dataset does not ship, and roughness 0 turned it into // a mirror where the schema's 0.5 is an ordinary surface. let fallback = float4(tk::TEX_FALLBACK) - .or_else(|| { - surface_input - .attribute() - .get_at::(eval_time()) - .ok() - .flatten() - .and_then(|v| sdf_float4(&v)) - }) + .or_else(|| value_at(surface_input.attribute()).and_then(decode_float4)) .or_else(|| preview_surface_default(name)) .unwrap_or([0.0, 0.0, 0.0, 1.0]); let loaded = load_uv_texture(&file, space, caches).map(crate::TextureRef); @@ -383,12 +371,7 @@ pub(super) fn preview_displacement( caches, ) else { // Already warned about; the input keeps its own constant. - let own = input - .attribute() - .get_at::(eval_time()) - .ok() - .flatten() - .and_then(|v| sdf_float4(&v))?; + let own = value_at(input.attribute()).and_then(decode_float4)?; return constant(own[0]); }; if uv.tex.is_none() { @@ -400,19 +383,14 @@ pub(super) fn preview_displacement( .value_producing_attributes(ProducerFilter::Any) .ok() .and_then(|p| p.into_iter().next()) - .and_then(|a| { - a.attribute() - .get_at::(eval_time()) - .ok() - .flatten() - }) - .and_then(|v| sdf_float4(&v))?; + .and_then(|a| value_at(a.attribute())) + .and_then(decode_float4)?; let c = value[0]; (c != 0.0 && c.is_finite()).then(|| Displacement::new(DisplacementValue::Constant(c))) } /// A `UsdPreviewSurface` input's schema default, widened to four channels the -/// way [`sdf_float4`] widens an authored value — what the input reads when a +/// way [`decode_float4`] widens an authored value — what the input reads when a /// texture drives it, the texture fails, and nothing else was authored. /// Values from the UsdPreviewSurface specification. fn preview_surface_default(input: &str) -> Option<[f32; 4]> { @@ -435,22 +413,6 @@ fn preview_surface_default(input: &str) -> Option<[f32; 4]> { }) } -/// A shading value widened to four channels, the shape `UsdUVTexture`'s -/// `scale`/`bias`/`fallback` have: a `float4` as authored, a colour with -/// alpha 1, a scalar in every channel. -fn sdf_float4(v: &sdf::Value) -> Option<[f32; 4]> { - Some(match v { - sdf::Value::Vec4f(v) => [v.x, v.y, v.z, v.w], - sdf::Value::Vec4d(v) => [v.x as f32, v.y as f32, v.z as f32, v.w as f32], - sdf::Value::Vec4h(v) => [v.x.to_f32(), v.y.to_f32(), v.z.to_f32(), v.w.to_f32()], - sdf::Value::Vec3f(v) => [v.x, v.y, v.z, 1.0], - sdf::Value::Vec3d(v) => [v.x as f32, v.y as f32, v.z as f32, 1.0], - sdf::Value::Float(f) => [*f; 4], - sdf::Value::Double(d) => [*d as f32; 4], - _ => return None, - }) -} - /// A `UsdPreviewSurface`'s constant inputs as an [`OpenPBR`]. A /// texture-connected input leaves its field at the default here; /// [`preview_surface_material`] is what drives it. diff --git a/crates/crust-core/src/scene/usd_import/products.rs b/crates/crust-core/src/scene/usd_import/products.rs index 0c94e33b..33a33884 100644 --- a/crates/crust-core/src/scene/usd_import/products.rs +++ b/crates/crust-core/src/scene/usd_import/products.rs @@ -28,10 +28,9 @@ use tracing::{debug, warn}; use crate::aov::{Accumulation, AovProduct, AovRequest, AovSource, AovVar, Precision}; -use super::attrs::{custom_bool, custom_token}; +use super::attrs::{custom_bool, custom_token, decode_number, prim_value, value_at}; use super::prim_at; use super::settings::render_settings_path; -use super::time::eval_time; const DRIVER_PARAMETERS: &str = "driver:parameters:"; @@ -58,11 +57,7 @@ struct Base { } fn read_resolution(view: &impl RenderSettingsBase) -> Option<(usize, usize)> { - let v = view - .resolution_attr() - .get_at::(eval_time()) - .ok()??; - let v = v.try_as_vec_2i()?; + let v = value_at(&view.resolution_attr())?.try_as_vec_2i()?; (v.x > 0 && v.y > 0).then_some((v.x as usize, v.y as usize)) } @@ -247,12 +242,7 @@ fn describe_resolution(resolution: Option<(usize, usize)>) -> String { /// every one of them at its fallback, and those need no word. fn warn_unhonoured(prim: &Prim) { let mut ignored = Vec::new(); - let value = |name: &str| { - prim.attribute(name) - .get_at::(eval_time()) - .ok() - .flatten() - }; + let value = |name: &str| prim_value(prim, name); if let Some(sdf::Value::Float(a)) = value("pixelAspectRatio") && a != 1.0 { @@ -312,17 +302,7 @@ fn driver_attributes(prim: &Prim) -> Vec<(String, String)> { /// A float attribute authored as any numeric type — Houdini writes /// `clearValue` as a float, hand-written files as an int. fn custom_number(prim: &Prim, name: &str) -> Option { - match prim - .attribute(name) - .get_at::(eval_time()) - .ok()?? - { - sdf::Value::Float(f) => Some(f), - sdf::Value::Double(d) => Some(d as f32), - sdf::Value::Half(h) => Some(h.to_f32()), - sdf::Value::Int(i) => Some(i as f32), - _ => None, - } + prim_value(prim, name).and_then(decode_number) } /// Components and precision of an Sdf type name or a Houdini diff --git a/crates/crust-core/src/scene/usd_import/settings.rs b/crates/crust-core/src/scene/usd_import/settings.rs index 07c041d9..b76c8057 100644 --- a/crates/crust-core/src/scene/usd_import/settings.rs +++ b/crates/crust-core/src/scene/usd_import/settings.rs @@ -11,9 +11,8 @@ use crate::filter::PixelFilter; use crate::light::LightSelection; use crate::tracer::{RenderSettings, SamplingStrategy}; -use super::attrs::{custom_bool, custom_f32, custom_i32, custom_token}; +use super::attrs::{custom_bool, custom_f32, custom_i32, custom_token, value_at}; use super::prim_at; -use super::time::eval_time; const DEFAULT_SPP: u32 = 128; const DEFAULT_MAX_DEPTH: u32 = 32; @@ -148,9 +147,7 @@ pub(super) fn import_render_settings(stage: &Stage) -> RenderSettings { }; let (mut w, mut h) = (DEFAULT_WIDTH, DEFAULT_HEIGHT); - if let Ok(Some(v)) = s.resolution_attr().get_at::(eval_time()) - && let Some(v2) = v.try_as_vec_2i() - { + if let Some(v2) = value_at(&s.resolution_attr()).and_then(|v| v.try_as_vec_2i()) { w = v2.x as usize; h = v2.y as usize; } diff --git a/crates/crust-core/src/scene/usd_import/shapes.rs b/crates/crust-core/src/scene/usd_import/shapes.rs index 62314f64..76af3cad 100644 --- a/crates/crust-core/src/scene/usd_import/shapes.rs +++ b/crates/crust-core/src/scene/usd_import/shapes.rs @@ -4,8 +4,6 @@ use std::sync::Arc; use crust_rt::{CubicCurveSegment, CurveSegment, Geometry, SceneBuilder as RtSceneBuilder}; use glam::{Affine3A, Mat4 as GMat4, Vec3, Vec3A}; -use openusd::gf::Vec3f; -use openusd::sdf; use openusd::usd::Prim; use openusd_schemas::geom::{ BasisCurves as UsdBasisCurves, Curves as UsdCurves, PointBased, Sphere as UsdSphere, @@ -15,8 +13,10 @@ use tracing::{debug, warn}; use crate::material::Material; use crate::rt_world::WorldBuilder; -use super::attrs::{custom_token, prim_motion_translate, prim_ray_mask}; -use super::time::eval_time; +use super::attrs::{ + attr_f32, custom_token, decode_f32_array, decode_i32_array, decode_vec3f_array, + prim_motion_translate, prim_ray_mask, value_at, +}; // ----------------------------------------------------------------------- // Sphere @@ -24,17 +24,7 @@ use super::time::eval_time; /// The authored `radius`, defaulting to USD's 1.0. pub(super) fn sphere_radius(sphere: &UsdSphere) -> f32 { - sphere - .radius_attr() - .get_at::(eval_time()) - .ok() - .flatten() - .and_then(|v| match v { - sdf::Value::Double(d) => Some(d as f32), - sdf::Value::Float(f) => Some(f), - _ => None, - }) - .unwrap_or(1.0) + attr_f32(&sphere.radius_attr()).unwrap_or(1.0) } pub(super) fn emit_sphere( @@ -181,24 +171,8 @@ pub(super) fn curve_segments( prim: &Prim, curves: &UsdBasisCurves, ) -> Option<(Vec, Vec)> { - let points: Option> = curves - .points_attr() - .get_at::(eval_time()) - .ok() - .flatten() - .and_then(|v| match v { - sdf::Value::Vec3fVec(v) => Some(v), - _ => None, - }); - let counts: Option> = curves - .curve_vertex_counts_attr() - .get_at::(eval_time()) - .ok() - .flatten() - .and_then(|v| match v { - sdf::Value::IntVec(v) => Some(v), - _ => None, - }); + let points = value_at(&curves.points_attr()).and_then(decode_vec3f_array); + let counts = value_at(&curves.curve_vertex_counts_attr()).and_then(decode_i32_array); let (points, counts) = match (points, counts) { (Some(p), Some(c)) => (p, c), _ => { @@ -211,15 +185,8 @@ pub(super) fn curve_segments( }; let pts: Vec = points.iter().map(|p| Vec3A::new(p.x, p.y, p.z)).collect(); - let widths: Vec = curves - .widths_attr() - .get_at::(eval_time()) - .ok() - .flatten() - .and_then(|v| match v { - sdf::Value::FloatVec(v) => Some(v), - _ => None, - }) + let widths: Vec = value_at(&curves.widths_attr()) + .and_then(decode_f32_array) .unwrap_or_else(|| vec![1.0]); // USD defaults: type = cubic, basis = bezier. diff --git a/crates/crust-core/src/scene/usd_import/xform.rs b/crates/crust-core/src/scene/usd_import/xform.rs index 5d14ccd6..d0408549 100644 --- a/crates/crust-core/src/scene/usd_import/xform.rs +++ b/crates/crust-core/src/scene/usd_import/xform.rs @@ -10,7 +10,8 @@ use openusd_schemas::geom::{ use openusd_schemas::lux::{RectLight, SphereLight}; use tracing::warn; -use super::time::{eval_time, xform_time}; +use super::attrs::{decode_f32, decode_vec3, prim_value}; +use super::time::xform_time; /// USD authors 4x4 matrices as row-vector row-major (translation in the /// last row, indices 12..15). glam::Mat4 is column-major with the @@ -50,7 +51,7 @@ fn usd_mat_to_glam(m: Matrix4d) -> GMat4 { /// floating objects against sky. Stacks with an op we cannot decode fall /// back to openusd's composition with a warning, so unusual scenes behave /// no worse than before. -pub(super) fn local_matrix_at(stage: &Stage, prim: &Prim) -> GMat4 { +fn local_matrix_at(stage: &Stage, prim: &Prim) -> GMat4 { match compose_xform_ops(prim) { Some(m) => m, None => { @@ -76,12 +77,9 @@ pub(super) fn local_matrix_at(stage: &Stage, prim: &Prim) -> GMat4 { /// /// Returns `None` if any op token or value cannot be decoded. fn compose_xform_ops(prim: &Prim) -> Option { - let order = match prim - .attribute("xformOpOrder") - .get_at::(eval_time()) - { - Ok(Some(sdf::Value::TokenVec(order))) => order, - Ok(Some(_)) => return None, + let order = match prim_value(prim, "xformOpOrder") { + Some(sdf::Value::TokenVec(order)) => order, + Some(_) => return None, // No order authored: authored xformOp attrs (if any) do not apply. _ => return Some(GMat4::IDENTITY), }; @@ -113,15 +111,13 @@ fn xform_op_matrix(prim: &Prim, name: &str) -> Option { // Suffixes name op instances (`xformOp:translate:pivot`); the kind is // the first segment. let kind = kind.split(':').next().unwrap_or(kind); - let value = prim - .attribute(name) - .get_at::(eval_time()) - .ok() - .flatten()?; + let value = prim_value(prim, name)?; + let vec3 = |v: sdf::Value| decode_vec3(v).map(Vec3::from); + let degrees = |v: sdf::Value| decode_f32(v).map(f32::to_radians); match kind { - "translate" => Some(GMat4::from_translation(value_as_vec3(&value)?)), - "scale" => Some(GMat4::from_scale(value_as_vec3(&value)?)), + "translate" => Some(GMat4::from_translation(vec3(value)?)), + "scale" => Some(GMat4::from_scale(vec3(value)?)), "transform" => match value { sdf::Value::Matrix4d(m) => Some(usd_mat_to_glam(m)), _ => None, @@ -135,14 +131,14 @@ fn xform_op_matrix(prim: &Prim, name: &str) -> Option { )), _ => None, }, - "rotateX" => Some(GMat4::from_rotation_x(value_as_f32(&value)?.to_radians())), - "rotateY" => Some(GMat4::from_rotation_y(value_as_f32(&value)?.to_radians())), - "rotateZ" => Some(GMat4::from_rotation_z(value_as_f32(&value)?.to_radians())), + "rotateX" => Some(GMat4::from_rotation_x(degrees(value)?)), + "rotateY" => Some(GMat4::from_rotation_y(degrees(value)?)), + "rotateZ" => Some(GMat4::from_rotation_z(degrees(value)?)), // Euler triples: the vector components are always the X/Y/Z-axis // angles in degrees; the op name gives the application order, first // named axis applied to the point first (so it sits rightmost). "rotateXYZ" | "rotateXZY" | "rotateYXZ" | "rotateYZX" | "rotateZXY" | "rotateZYX" => { - let v = value_as_vec3(&value)?; + let v = vec3(value)?; let rx = GMat4::from_rotation_x(v.x.to_radians()); let ry = GMat4::from_rotation_y(v.y.to_radians()); let rz = GMat4::from_rotation_z(v.z.to_radians()); @@ -159,24 +155,6 @@ fn xform_op_matrix(prim: &Prim, name: &str) -> Option { } } -fn value_as_vec3(value: &sdf::Value) -> Option { - match value { - sdf::Value::Vec3f(v) => Some(Vec3::new(v.x, v.y, v.z)), - sdf::Value::Vec3d(v) => Some(Vec3::new(v.x as f32, v.y as f32, v.z as f32)), - sdf::Value::Vec3h(v) => Some(Vec3::new(v.x.to_f32(), v.y.to_f32(), v.z.to_f32())), - _ => None, - } -} - -fn value_as_f32(value: &sdf::Value) -> Option { - match value { - sdf::Value::Float(v) => Some(*v), - sdf::Value::Double(v) => Some(*v as f32), - sdf::Value::Half(v) => Some(v.to_f32()), - _ => None, - } -} - /// openusd's own composition, kept as the fallback for op stacks /// `compose_xform_ops` cannot decode. Known to compose multi-op stacks in /// the wrong order (see `local_matrix_at`). @@ -214,7 +192,20 @@ fn local_matrix_via_openusd(stage: &Stage, prim: &Prim) -> GMat4 { GMat4::IDENTITY } -pub(super) fn resets_xform_stack_at(stage: &Stage, prim: &Prim) -> bool { +/// `prim`'s transform given its parent's: `parent · local`, or `local` alone +/// when the prim authors `!resetXformStack!` — the one composition rule every +/// walk (the traversal, the placement count, the prototype walk, a camera's +/// ancestor chain) applies. +pub(super) fn compose_with_parent(stage: &Stage, prim: &Prim, parent: GMat4) -> GMat4 { + let local = local_matrix_at(stage, prim); + if resets_xform_stack_at(stage, prim) { + local + } else { + parent * local + } +} + +fn resets_xform_stack_at(stage: &Stage, prim: &Prim) -> bool { if let Ok(Some(x)) = Xform::get(stage, prim.path().clone()) { return x.resets_xform_stack().unwrap_or(false); } diff --git a/crates/crust-render/examples/maketx.rs b/crates/crust-render/examples/maketx.rs index 1cdced25..61f9f357 100644 --- a/crates/crust-render/examples/maketx.rs +++ b/crates/crust-render/examples/maketx.rs @@ -99,17 +99,7 @@ fn main() { let mut jobs: Vec = Vec::new(); if input.contains("") || input.contains("") { - for v in 0..10u32 { - for u in 0..10u32 { - let name = input - .replace("", &(1001 + u + 10 * v).to_string()) - .replace("", &format!("u{}_v{}", u + 1, v + 1)); - let p = PathBuf::from(&name); - if p.exists() { - jobs.push(p); - } - } - } + jobs = crust_assets::texture_files(Path::new(&input)); if jobs.is_empty() { eprintln!("no tiles of {input} found on disk"); std::process::exit(1); From 51378e5dabda9ea9d592cae3cb20c0e313f003c6 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 5 Oct 2026 06:52:53 +0000 Subject: [PATCH 2/5] Integrator and fixtures: pay off duplicated logic - russian_roulette(): the survival test the surface bounce, the region phase scatter and the carried-medium scatter each pasted. - utils::exp3 replaces six per-component exp copies. - TRACE_T_MIN (crust_core::ray) names the 0.001 every trace asks for. - surface_visibility(): the one shadow-ray visibility (cutouts included) NEE and the learned light cache's training share, replacing the cache's own copy of the occluded/cutout logic. - LightList::pick_index_at forced inline: LLVM outlined it once trace_path grew, +0.6% instructions on cornellbox; inlined, the render runs 0.01% fewer than before this change (callgrind, 2 spp). - crust-rt: fixtures/mod.rs holds the uv_sphere, the three scenes and the ray batch ray_throughput, traversal_probe and the criterion bench shared by copy; their LCGs become openqmc::pcg::Rng. ray_throughput now reads --layout before building the default scenes. - mtlx_bench and jit_bench share bench_common (texture, points, A/B timer, material walk). All sample scenes bit-identical at 16 spp. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_019nwge6NCTPhuk1VPuZucRF --- crates/crust-core/src/lib.rs | 4 +- crates/crust-core/src/light/list.rs | 6 +- crates/crust-core/src/light_cache.rs | 27 +-- crates/crust-core/src/material/closure/mx.rs | 7 +- crates/crust-core/src/medium.rs | 2 +- crates/crust-core/src/ray.rs | 11 + crates/crust-core/src/subsurface.rs | 17 +- crates/crust-core/src/tracer/mod.rs | 2 +- crates/crust-core/src/tracer/path.rs | 201 +++++++++--------- crates/crust-core/src/volume.rs | 2 +- crates/crust-jit/examples/jit_bench.rs | 112 ++-------- .../crust-mtlx/examples/bench_common/mod.rs | 87 ++++++++ crates/crust-mtlx/examples/mtlx_bench.rs | 107 ++-------- .../crust-render/examples/light_occlusion.rs | 18 +- crates/crust-rt/benches/traversal.rs | 132 ++---------- crates/crust-rt/examples/ray_throughput.rs | 158 +++----------- crates/crust-rt/examples/traversal_probe.rs | 52 ++--- crates/crust-rt/fixtures/mod.rs | 129 +++++++++++ crates/utils/src/common.rs | 7 + crates/utils/src/lib.rs | 2 +- 20 files changed, 476 insertions(+), 607 deletions(-) create mode 100644 crates/crust-mtlx/examples/bench_common/mod.rs create mode 100644 crates/crust-rt/fixtures/mod.rs diff --git a/crates/crust-core/src/lib.rs b/crates/crust-core/src/lib.rs index 3744b467..dd8103c8 100644 --- a/crates/crust-core/src/lib.rs +++ b/crates/crust-core/src/lib.rs @@ -93,7 +93,9 @@ pub use lux::{ pub use material::*; pub use medium::Medium; pub use pdf::{InvPdfArea, PdfSolidAngle}; -pub use ray::{MASK_ALL, MASK_CAMERA, MASK_INDIRECT, MASK_SHADOW, Ray, RayCone, RayMask}; +pub use ray::{ + MASK_ALL, MASK_CAMERA, MASK_INDIRECT, MASK_SHADOW, Ray, RayCone, RayMask, TRACE_T_MIN, +}; pub use rt_world::{FaceMap, FanSlice, SubFace, UvMap, World, WorldBuilder, WorldHit, tangent_of}; pub use scene::Scene; pub use scene::{AssetLoader, NoAssets, UsdImportOptions}; diff --git a/crates/crust-core/src/light/list.rs b/crates/crust-core/src/light/list.rs index ba2b6256..9ae88258 100644 --- a/crates/crust-core/src/light/list.rs +++ b/crates/crust-core/src/light/list.rs @@ -432,7 +432,11 @@ impl LightList { } /// [`LightList::pick_at`] as an index into [`LightList::lights`]. - #[inline] + /// + /// Forced inline: it sits on every NEE pick, and LLVM's own threshold + /// outlined it once `trace_path` grew by a few instructions elsewhere, + /// which cost cornellbox 0.6% of its instructions (callgrind, 2 spp). + #[inline(always)] pub fn pick_index_at(&self, p: Vec3A, u: f32) -> Option<(usize, f32)> { match self.cache.as_ref().and_then(|c| c.lookup(p)) { Some((pmf, cdf)) => { diff --git a/crates/crust-core/src/light_cache.rs b/crates/crust-core/src/light_cache.rs index e2fa69f9..bfd6e51d 100644 --- a/crates/crust-core/src/light_cache.rs +++ b/crates/crust-core/src/light_cache.rs @@ -198,7 +198,8 @@ pub(crate) fn train( } let mut ray = camera.get_ray(u, v, [cam[2], cam[3]], 0.0); for depth in 0..=TRAIN_BOUNCES { - let Some(hit) = world.intersect(&ray, 0.001, f32::INFINITY) else { + let Some(hit) = world.intersect(&ray, crate::ray::TRACE_T_MIN, f32::INFINITY) + else { break; }; let vertex = root.new_domain(1 + depth as i32); @@ -243,22 +244,14 @@ pub(crate) fn train( // The integrator's visibility, cutouts included: // a light seen through a leaf card is trained at // the share the card lets through. - let t_max = crate::tracer::shadow_t_max(ls.distance); - let mut through = 1.0; - if world.occluded(&shadow, 0.001, t_max) { - through = if world.has_cutouts() { - crate::tracer::cutout_through( - world, - &shadow, - t_max, - &mut crate::stats::RayStats::default(), - ) - } else { - 0.0 - }; - if through == 0.0 { - continue; - } + let through = crate::tracer::surface_visibility( + world, + &shadow, + ls.distance, + &mut crate::stats::RayStats::default(), + ); + if through == 0.0 { + continue; } let e = through * lights.luma().of(c) / ls.pdf.get(); if e.is_finite() { diff --git a/crates/crust-core/src/material/closure/mx.rs b/crates/crust-core/src/material/closure/mx.rs index a3dcd348..0f7a8cfb 100644 --- a/crates/crust-core/src/material/closure/mx.rs +++ b/crates/crust-core/src/material/closure/mx.rs @@ -344,9 +344,10 @@ fn eval_sensitivity(opd: f32, shift: Vec3A) -> Vec3A { let pos = Vec3A::new(1.6810e+06, 1.7953e+06, 2.2084e+06); let var = Vec3A::new(4.3278e+09, 9.3046e+09, 6.6121e+09); let cos = |v: Vec3A| Vec3A::new(v.x.cos(), v.y.cos(), v.z.cos()); - let exp = |v: Vec3A| Vec3A::new(v.x.exp(), v.y.exp(), v.z.exp()); - let mut xyz = - val * (2.0 * PI * var).powf(0.5) * cos(pos * phase + shift) * exp(-var * phase * phase); + let mut xyz = val + * (2.0 * PI * var).powf(0.5) + * cos(pos * phase + shift) + * utils::exp3(-var * phase * phase); xyz.x += 9.7470e-14 * (2.0 * PI * 4.5282e+09f32).sqrt() * (2.2399e+06 * phase + shift.x).cos() diff --git a/crates/crust-core/src/medium.rs b/crates/crust-core/src/medium.rs index 33debfb2..3c9a7830 100644 --- a/crates/crust-core/src/medium.rs +++ b/crates/crust-core/src/medium.rs @@ -124,7 +124,7 @@ impl Medium { /// Beer–Lambert transmittance across a segment of length `t`. pub fn transmittance(&self, t: f32) -> Vec3A { let e = (self.sigma_a + self.sigma_s) * t; - Vec3A::new((-e.x).exp(), (-e.y).exp(), (-e.z).exp()) + utils::exp3(-e) } /// True when the medium scatters (subsurface, participating volumes). diff --git a/crates/crust-core/src/ray.rs b/crates/crust-core/src/ray.rs index df499821..df7648dd 100644 --- a/crates/crust-core/src/ray.rs +++ b/crates/crust-core/src/ray.rs @@ -3,6 +3,17 @@ use glam::Vec3A; pub use crust_rt::{MASK_ALL, MASK_CAMERA, MASK_INDIRECT, MASK_SHADOW, RayMask}; +/// The near bound every trace in the renderer asks for: `(TRACE_T_MIN, ∞)` +/// for a closest hit, `(TRACE_T_MIN, t_max)` for a shadow ray — the offset +/// that keeps a ray from re-hitting the surface it leaves. +/// +/// A constant, not a parameter, on purpose. With every caller passing it, +/// LLVM propagates it into the kernel; one caller passing a variable was +/// enough to lose that, and cornellbox ran 0.3% more instructions in the +/// triangle test. A ray that must start elsewhere (the subsurface walk, a +/// segment restarted past a cutout) moves its origin instead. +pub const TRACE_T_MIN: f32 = 0.001; + /// The ray's texture-filtering footprint, as a cone about its axis. /// /// This is the renderer's answer to "how much texture does this ray cover", diff --git a/crates/crust-core/src/subsurface.rs b/crates/crust-core/src/subsurface.rs index daad400c..69a87397 100644 --- a/crates/crust-core/src/subsurface.rs +++ b/crates/crust-core/src/subsurface.rs @@ -36,14 +36,14 @@ use glam::Vec3A; use std::f32::consts::{FRAC_1_PI, PI}; -use utils::cosine_hemisphere; +use utils::{cosine_hemisphere, exp3}; use crate::PathSampler; use crate::hittable::HitRecord; use crate::material::brdf::tangent_frame; use crate::material::{Material, ScatterSample}; use crate::medium::{hg_phase, sample_henyey_greenstein}; -use crate::ray::{MASK_ALL, Ray}; +use crate::ray::{MASK_ALL, Ray, TRACE_T_MIN}; use crate::rt_world::World; /// Walk steps before a walk is given up as absorbed (Typhoon, Cycles). @@ -228,19 +228,6 @@ pub fn backward_dwivedi_fraction(opposite: f32, from_entry: f32, nu: f32) -> f32 1.0 / (1.0 + ((opposite - 2.0 * d) / nu).exp()) } -fn exp3(v: Vec3A) -> Vec3A { - Vec3A::new(v.x.exp(), v.y.exp(), v.z.exp()) -} - -/// The interval every other `World::intersect` in the renderer asks for. -/// -/// The walk asks for the same one and moves the ray's origin instead of its -/// bounds. With every caller passing `(0.001, ∞)`, LLVM propagates both -/// constants into the kernel; one caller passing variables was enough to -/// lose that, and cornellbox — which never walks — ran 0.3% more -/// instructions in the triangle test. -const TRACE_T_MIN: f32 = 0.001; - /// Closest hit on `owner` from `pos` along the unit `dir` within /// `(t_min, t_max)`, stepping past every other geometry. The record's `t` /// is measured from `pos`. diff --git a/crates/crust-core/src/tracer/mod.rs b/crates/crust-core/src/tracer/mod.rs index 9de769b8..793c713f 100644 --- a/crates/crust-core/src/tracer/mod.rs +++ b/crates/crust-core/src/tracer/mod.rs @@ -14,7 +14,7 @@ use crate::volume::Volumes; use crate::{LightList, LightSelection, PathSampler}; mod path; -pub(crate) use path::{cutout_through, shadow_t_max}; +pub(crate) use path::surface_visibility; mod route; mod settings; diff --git a/crates/crust-core/src/tracer/path.rs b/crates/crust-core/src/tracer/path.rs index 682c9ddc..652e1cb8 100644 --- a/crates/crust-core/src/tracer/path.rs +++ b/crates/crust-core/src/tracer/path.rs @@ -5,6 +5,7 @@ //! `escaped_emission`); change both or neither. use glam::Vec3A; +use utils::exp3; use crate::aov::FirstHit; use crate::guiding::SampleData; @@ -13,7 +14,7 @@ use crate::material::{Material, ScatterSample, ShadingPoint}; use crate::medium::sample_henyey_greenstein; use crate::pdf::PdfSolidAngle; use crate::profile::Section; -use crate::ray::{Ray, RayMask}; +use crate::ray::{Ray, RayMask, TRACE_T_MIN}; use crate::rt_world::{World, WorldHit}; use crate::stats::RayStats; use crate::subsurface::{ExitLambertian, WalkCost, random_walk}; @@ -56,6 +57,37 @@ const TRAIN_RADIANCE_CLAMP: f32 = 1e3; const RR_START_BOUNCE: usize = 3; const RR_MIN_PROB: f32 = 0.05; +/// Russian roulette at vertex `depth`. Past [`RR_START_BOUNCE`] the path +/// survives with a probability tracking its throughput `beta`, floored at +/// [`RR_MIN_PROB`], and a survivor's `beta` and continuation `factor` are +/// divided by it. Returns that probability — 1 when the vertex is not tested +/// or survival is certain — or `None` when the path is killed, leaving both +/// untouched. The one roulette every vertex kind (surface bounce, region +/// phase scatter, carried-medium scatter) applies. +#[inline(always)] +fn russian_roulette( + depth: usize, + v: PathSampler, + beta: &mut Vec3A, + factor: &mut Vec3A, + stats: &mut RayStats, +) -> Option { + if depth < RR_START_BOUNCE { + return Some(1.0); + } + stats.rr_tested += 1; + let p_survive = beta.max_element().clamp(RR_MIN_PROB, 1.0); + if p_survive < 1.0 { + if v.new_domain(K_RR).draw_rnd_f32::<1>()[0] >= p_survive { + stats.rr_killed += 1; + return None; + } + *factor /= p_survive; + *beta /= p_survive; + } + Some(p_survive) +} + pub fn ray_color( r: &Ray, world: &World, @@ -604,21 +636,53 @@ fn shadow_transmittance( // early-exit traversal beats searching for the closest hit. let _p = profile::scope_if::(Section::Occlusion); stats.shadow_rays += 1; - if world.occluded(shadow_ray, 0.001, shadow_t_max(distance)) { - // Blocked — unless only cutouts block it, which the any-hit query - // cannot tell apart. An open segment crosses no cutout either, so it - // keeps the fast answer. - if !world.has_cutouts() { - stats.shadow_occluded += 1; - return Vec3A::ZERO; - } - return cutout_shadow(world, volumes, shadow_ray, distance, vertex, stats); + let through = surface_visibility(world, shadow_ray, distance, stats); + if through == 0.0 { + stats.shadow_occluded += 1; + return Vec3A::ZERO; } if volumes.is_empty() { - return Vec3A::ONE; + return Vec3A::splat(through); } let mut rng = vertex.new_domain(K_NEE_SHADOW).rng(); - volumes.transmittance(shadow_ray, 0.001, distance - 0.001, &mut rng) + through * volumes.transmittance(shadow_ray, TRACE_T_MIN, distance - TRACE_T_MIN, &mut rng) +} + +/// How much of a shadow ray toward a light sample `distance` away surfaces +/// let through: 1 when nothing blocks it, 0 when an opaque surface does, and +/// `Π (1 − opacity)` over the cutouts it crosses otherwise ([`cutout_through`], +/// deterministic where the bounce side is stochastic — [`pass_cutouts`] — so +/// the product is the lower-variance estimate of the same visibility). +/// +/// The one answer to "does this light reach here" that NEE +/// ([`shadow_transmittance`]) and the learned light cache's training share: +/// the cache must train on the visibility the integrator renders with. +#[inline] +pub(crate) fn surface_visibility( + world: &World, + ray: &Ray, + distance: f32, + stats: &mut RayStats, +) -> f32 { + let t_max = shadow_t_max(distance); + if !world.occluded(ray, TRACE_T_MIN, t_max) { + return 1.0; + } + // Blocked — unless only cutouts block it, which the any-hit query cannot + // tell apart. An open segment crosses no cutout either, so it keeps the + // fast answer. + if !world.has_cutouts() { + return 0.0; + } + cutout_visibility(world, ray, t_max, stats) +} + +/// [`cutout_through`] out of line: only a blocked shadow ray in a world with +/// cutouts reaches it. +#[cold] +#[inline(never)] +fn cutout_visibility(world: &World, ray: &Ray, t_max: f32, stats: &mut RayStats) -> f32 { + cutout_through(world, ray, t_max, stats) } /// How many cutouts one segment is followed through, on either side: past @@ -657,7 +721,7 @@ fn restarted(ray: &Ray, t: f32) -> Ray { /// so a far hit is not met again through rounding. #[inline] fn resume_before(t: f32) -> f32 { - t - 0.001 + t.abs().max(1.0) * 1e-5 + t - TRACE_T_MIN + t.abs().max(1.0) * 1e-5 } /// Where a shadow ray toward a light sample `distance` away stops: short of @@ -674,37 +738,8 @@ fn resume_before(t: f32) -> f32 { /// Surfaces only: volume transmittance has no light surface to stop short of /// and keeps `distance − 0.001`, which reaches the light at any distance. #[inline] -pub(crate) fn shadow_t_max(distance: f32) -> f32 { - (distance - 0.001).min(distance * (1.0 - 1e-6)) -} - -/// [`shadow_transmittance`] for a shadow ray the any-hit query found blocked -/// in a world with cutouts: `Π (1 − opacity)` over every hit up to the light, -/// or zero at the first hit on a material without a cutout, times the volume -/// transmittance. Deterministic where the bounce side is stochastic -/// ([`pass_cutouts`]): both estimate the same visibility, and the product is -/// the lower-variance of the two. -#[cold] -#[inline(never)] -fn cutout_shadow( - world: &World, - volumes: &Volumes, - ray: &Ray, - distance: f32, - vertex: PathSampler, - stats: &mut RayStats, -) -> Vec3A { - let t_max = shadow_t_max(distance); - let through = cutout_through(world, ray, t_max, stats); - if through == 0.0 { - stats.shadow_occluded += 1; - return Vec3A::ZERO; - } - if volumes.is_empty() { - return Vec3A::splat(through); - } - let mut rng = vertex.new_domain(K_NEE_SHADOW).rng(); - through * volumes.transmittance(ray, 0.001, distance - 0.001, &mut rng) +fn shadow_t_max(distance: f32) -> f32 { + (distance - TRACE_T_MIN).min(distance * (1.0 - 1e-6)) } /// The fraction of the segment `(0.001, t_max)` of `ray` that cutouts let @@ -714,14 +749,13 @@ fn cutout_shadow( /// [`pass_cutouts`] keeps, which treats the hit past its last crossing as /// present, so a stack exactly that deep is clear on both sides. /// -/// Shared by NEE ([`cutout_shadow`]) and the learned light cache's training, -/// whose shadow rays must see the visibility the integrator does. -pub(crate) fn cutout_through(world: &World, ray: &Ray, t_max: f32, stats: &mut RayStats) -> f32 { +/// Reached through [`surface_visibility`]. +fn cutout_through(world: &World, ray: &Ray, t_max: f32, stats: &mut RayStats) -> f32 { let (mut t, mut segment) = (0.0, ray.clone()); let mut kept = 1.0; for crossing in 0..=MAX_CUTOUT_CROSSINGS { stats.cutout_rays += 1; - let hit = world.intersect(&segment, 0.001, f32::INFINITY); + let hit = world.intersect(&segment, TRACE_T_MIN, f32::INFINITY); let Some(h) = hit.filter(|h| t + h.rec.t < t_max) else { return kept; }; @@ -796,7 +830,7 @@ fn pass_cutouts<'w>( stats.cutout_rays += 1; let t = resume_before(h.rec.t); *hit = world - .intersect(&restarted(ray, t), 0.001, f32::INFINITY) + .intersect(&restarted(ray, t), TRACE_T_MIN, f32::INFINITY) .map(|mut next| { next.rec.t += t; next @@ -964,7 +998,7 @@ pub(super) fn trace_path( stats.closest_hit += 1; let mut hit = { let _p = profile::scope_if::(Section::Trace); - world.intersect(&ray, 0.001, f32::INFINITY) + world.intersect(&ray, TRACE_T_MIN, f32::INFINITY) }; let cut0 = stats.cutout_passes; if world.has_cutouts() { @@ -982,7 +1016,8 @@ pub(super) fn trace_path( } if !volumes.is_empty() { let mut rng = v.new_domain(K_VOLUME).rng(); - emitted *= volumes.transmittance(&ray, 0.001, hit.rec.t, &mut rng); + emitted *= + volumes.transmittance(&ray, TRACE_T_MIN, hit.rec.t, &mut rng); } let last = records.last_mut().expect("prev implies a record"); last.next_emit = emitted; @@ -1003,7 +1038,7 @@ pub(super) fn trace_path( } else { stats.closest_hit += 1; let _p = profile::scope_if::(Section::Trace); - world.intersect(&ray, 0.001, f32::INFINITY) + world.intersect(&ray, TRACE_T_MIN, f32::INFINITY) }; // Patched in place, on the cold side only: an `if` that yields the // hit from either arm copies all of it at every vertex (+0.8% of @@ -1048,7 +1083,7 @@ pub(super) fn trace_path( } else { let _p = profile::scope_if::(Section::Volume); let mut rng = v.new_domain(K_VOLUME).rng(); - volumes.sample_interaction(&ray, 0.001, t_lim, &mut rng) + volumes.sample_interaction(&ray, TRACE_T_MIN, t_lim, &mut rng) }; let (vol_tr, vol_emit) = match event { @@ -1099,20 +1134,11 @@ pub(super) fn trace_path( train: None, }; beta *= weight; - let mut survived = true; - if records.len() >= RR_START_BOUNCE { - stats.rr_tested += 1; - let p_survive = beta.max_element().clamp(RR_MIN_PROB, 1.0); - if p_survive < 1.0 { - if v.new_domain(K_RR).draw_rnd_f32::<1>()[0] >= p_survive { - survived = false; - stats.rr_killed += 1; - vrec.factor = Vec3A::ZERO; - } else { - vrec.factor /= p_survive; - beta /= p_survive; - } - } + let survived = + russian_roulette(records.len(), v, &mut beta, &mut vrec.factor, stats) + .is_some(); + if !survived { + vrec.factor = Vec3A::ZERO; } if let Some(ctx) = routing { // `V`, then the light NEE picked — the same draw @@ -1188,7 +1214,7 @@ pub(super) fn trace_path( // old code's `factor = albedo` with an extra Beer-Lambert on // top double-counted extinction. let e = (Vec3A::splat(sigma_bar) - (medium.sigma_a + medium.sigma_s)) * t_med; - let factor = medium.sigma_s / sigma_bar * Vec3A::new(e.x.exp(), e.y.exp(), e.z.exp()); + let factor = medium.sigma_s / sigma_bar * exp3(e); // Subsurface vertices run no NEE (their shadow rays are // blocked by the enclosing surface), so `prev = None` keeps // the next hit's emission at full weight — the pairing that @@ -1204,20 +1230,10 @@ pub(super) fn trace_path( train: None, }; beta *= vol_tr * factor; - let mut survived = true; - if records.len() >= RR_START_BOUNCE { - stats.rr_tested += 1; - let p_survive = beta.max_element().clamp(RR_MIN_PROB, 1.0); - if p_survive < 1.0 { - if v.new_domain(K_RR).draw_rnd_f32::<1>()[0] >= p_survive { - survived = false; - stats.rr_killed += 1; - vrec.factor = Vec3A::ZERO; - } else { - vrec.factor /= p_survive; - beta /= p_survive; - } - } + let survived = + russian_roulette(records.len(), v, &mut beta, &mut vrec.factor, stats).is_some(); + if !survived { + vrec.factor = Vec3A::ZERO; } if let Some(ctx) = routing { // A medium scatter is a `V` event; it runs no NEE. @@ -1313,7 +1329,7 @@ pub(super) fn trace_path( Some(m) if m.is_scattering() => { let sigma_bar = m.sigma_t_max().max(1e-4); let e = (Vec3A::splat(sigma_bar) - (m.sigma_a + m.sigma_s)) * rec.t; - Vec3A::new(e.x.exp(), e.y.exp(), e.z.exp()) + exp3(e) } Some(m) => m.transmittance(rec.t), None => Vec3A::ONE, @@ -1598,22 +1614,9 @@ pub(super) fn trace_path( // tracking the throughput, dividing it out on survival. Applies // to the whole continuation (bounce-hit emission included). beta *= atten * factor; - let mut survived = true; - let mut rr_survive = 1.0f32; - if records.len() >= RR_START_BOUNCE { - stats.rr_tested += 1; - let p_survive = beta.max_element().clamp(RR_MIN_PROB, 1.0); - if p_survive < 1.0 { - if v.new_domain(K_RR).draw_rnd_f32::<1>()[0] >= p_survive { - survived = false; - stats.rr_killed += 1; - } else { - factor /= p_survive; - beta /= p_survive; - rr_survive = p_survive; - } - } - } + let roulette = russian_roulette(records.len(), v, &mut beta, &mut factor, stats); + let survived = roulette.is_some(); + let rr_survive = roulette.unwrap_or(1.0); if survived { // Training samples cover continuous surface bounces only — diff --git a/crates/crust-core/src/volume.rs b/crates/crust-core/src/volume.rs index e3091d52..e60af151 100644 --- a/crates/crust-core/src/volume.rs +++ b/crates/crust-core/src/volume.rs @@ -543,7 +543,7 @@ impl Volumes { let mut tr = Vec3A::ONE; for &(i, a, b) in &spans { let e = self.regions[i].sigma_t_at_density(1.0) * (b - a); - tr *= Vec3A::new((-e.x).exp(), (-e.y).exp(), (-e.z).exp()); + tr *= utils::exp3(-e); } return tr; } diff --git a/crates/crust-jit/examples/jit_bench.rs b/crates/crust-jit/examples/jit_bench.rs index afd042d2..d5f98023 100644 --- a/crates/crust-jit/examples/jit_bench.rs +++ b/crates/crust-jit/examples/jit_bench.rs @@ -10,101 +10,35 @@ //! samples/MaterialXTeapotLion-1.0/Lion/Looks/lion_ldX.mtlx //! ``` +#[path = "../../crust-mtlx/examples/bench_common/mod.rs"] +mod bench_common; + +use bench_common::{for_each_material, time_ab}; use crust_jit::JitProgram; -use crust_mtlx::{Compiled, Doc, Host, ShadeCtx, Texture, TextureRef, Val, compile}; -use glam::Vec3A; -use std::hint::black_box; -use std::sync::Arc; use std::time::Instant; -struct Procedural; -impl Texture for Procedural { - fn eval(&self, u: f32, v: f32, width: f32) -> [f32; 4] { - [u.fract().abs(), v.fract().abs(), 0.5, 0.5 + width] - } -} - -fn procedural(_: &str, _: Option<&str>) -> Option { - Some(TextureRef(Arc::new(Procedural))) -} - -const POINTS: usize = 4096; -const REPEATS: usize = 15; - -fn points() -> Vec { - (0..POINTS) - .map(|i| { - let t = i as f32 / POINTS as f32; - ShadeCtx { - uv: (t * 2.0, (t * 17.0).fract()), - normal: Vec3A::new((t * 7.0).sin(), (t * 5.0).cos(), 1.0).normalize(), - tangent: Vec3A::X, - view: -Vec3A::new(0.2, (t * 3.0).sin(), 1.0).normalize(), - position: Vec3A::new(t, 1.0 - t, t * t), - uv_width: t * 0.01, - } - }) - .collect() -} - -/// Min-of-N nanoseconds per run for two evaluators, **interleaved**: each -/// repeat times both, alternating which goes first, so load or a frequency -/// change lands on both rather than on whichever phase it happened to hit -/// (the in-process version of `scripts/bench_ab.sh`). -fn time_ab( - pts: &[ShadeCtx], - mut a: impl FnMut(&ShadeCtx, &mut Vec), - mut b: impl FnMut(&ShadeCtx, &mut Vec), -) -> (f64, f64) { - let mut slots = Vec::new(); - let once = |run: &mut dyn FnMut(&ShadeCtx, &mut Vec), slots: &mut Vec| { - let t = Instant::now(); - for p in pts { - run(black_box(p), slots); - black_box(&*slots); - } - t.elapsed().as_nanos() as f64 / pts.len() as f64 - }; - let (mut best_a, mut best_b) = (f64::INFINITY, f64::INFINITY); - for rep in 0..REPEATS { - if rep % 2 == 0 { - best_a = best_a.min(once(&mut a, &mut slots)); - best_b = best_b.min(once(&mut b, &mut slots)); - } else { - best_b = best_b.min(once(&mut b, &mut slots)); - best_a = best_a.min(once(&mut a, &mut slots)); - } - } - (best_a, best_b) -} - fn main() { - let pts = points(); println!( "{:<36} {:>5} {:>12} {:>10} {:>10} {:>9}", "material", "ops", "inline/host", "ns/interp", "ns/jit", "compile" ); - for path in std::env::args().skip(1) { - let path = std::path::Path::new(&path); - let doc = Doc::open(path).expect("parse"); - for node in doc.by_category("surfacematerial") { - let mut c: Compiled = compile(path, Some(&node.name), &Host::new(&procedural)).unwrap(); - c.optimize(); - let p = &c.program; - let t0 = Instant::now(); - let jit = JitProgram::new(p).expect("jit"); - let compile_ms = t0.elapsed().as_secs_f64() * 1e3; - let (inl, host) = jit.split(); - let (t_i, t_j) = time_ab(&pts, |c, s| p.eval(c, s), |c, s| jit.eval(c, s)); - println!( - "{:<36} {:>5} {:>12} {:>10.1} {:>10.1} {:>7.2}ms", - node.name, - p.ops.len(), - format!("{inl}/{host}"), - t_i, - t_j, - compile_ms - ); - } - } + for_each_material(|name, compile, pts| { + let mut c = compile(); + c.optimize(); + let p = &c.program; + let t0 = Instant::now(); + let jit = JitProgram::new(p).expect("jit"); + let compile_ms = t0.elapsed().as_secs_f64() * 1e3; + let (inl, host) = jit.split(); + let (t_i, t_j) = time_ab(pts, |c, s| p.eval(c, s), |c, s| jit.eval(c, s)); + println!( + "{:<36} {:>5} {:>12} {:>10.1} {:>10.1} {:>7.2}ms", + name, + p.ops.len(), + format!("{inl}/{host}"), + t_i, + t_j, + compile_ms + ); + }); } diff --git a/crates/crust-mtlx/examples/bench_common/mod.rs b/crates/crust-mtlx/examples/bench_common/mod.rs new file mode 100644 index 00000000..92b4f863 --- /dev/null +++ b/crates/crust-mtlx/examples/bench_common/mod.rs @@ -0,0 +1,87 @@ +//! What `mtlx_bench` (crust-mtlx) and `jit_bench` (crust-jit) share, included +//! by each with `#[path]`: the procedural texture, the shading points, the +//! interleaved A/B timer and the walk over every `surfacematerial` of the files +//! given — so the two benches time the same work and differ only in what they +//! compare. + +use crust_mtlx::{Compiled, Doc, Host, ShadeCtx, Texture, TextureRef, Val, compile}; +use glam::Vec3A; +use std::hint::black_box; +use std::sync::Arc; +use std::time::Instant; + +pub struct Procedural; +impl Texture for Procedural { + fn eval(&self, u: f32, v: f32, width: f32) -> [f32; 4] { + [u.fract().abs(), v.fract().abs(), 0.5, 0.5 + width] + } +} + +pub fn procedural(_: &str, _: Option<&str>) -> Option { + Some(TextureRef(Arc::new(Procedural))) +} + +const POINTS: usize = 4096; +const REPEATS: usize = 15; + +pub fn points() -> Vec { + (0..POINTS) + .map(|i| { + let t = i as f32 / POINTS as f32; + ShadeCtx { + uv: (t * 2.0, (t * 17.0).fract()), + normal: Vec3A::new((t * 7.0).sin(), (t * 5.0).cos(), 1.0).normalize(), + tangent: Vec3A::X, + view: -Vec3A::new(0.2, (t * 3.0).sin(), 1.0).normalize(), + position: Vec3A::new(t, 1.0 - t, t * t), + uv_width: t * 0.01, + } + }) + .collect() +} + +/// Min-of-N nanoseconds per run for two evaluators, **interleaved**: each +/// repeat times both, alternating which goes first, so load or a frequency +/// change lands on both rather than on whichever phase it happened to hit +/// (the in-process version of `scripts/bench_ab.sh`). +pub fn time_ab( + pts: &[ShadeCtx], + mut a: impl FnMut(&ShadeCtx, &mut Vec), + mut b: impl FnMut(&ShadeCtx, &mut Vec), +) -> (f64, f64) { + let mut slots = Vec::new(); + let once = |run: &mut dyn FnMut(&ShadeCtx, &mut Vec), slots: &mut Vec| { + let t = Instant::now(); + for p in pts { + run(black_box(p), slots); + black_box(&*slots); + } + t.elapsed().as_nanos() as f64 / pts.len() as f64 + }; + let (mut best_a, mut best_b) = (f64::INFINITY, f64::INFINITY); + for rep in 0..REPEATS { + if rep % 2 == 0 { + best_a = best_a.min(once(&mut a, &mut slots)); + best_b = best_b.min(once(&mut b, &mut slots)); + } else { + best_b = best_b.min(once(&mut b, &mut slots)); + best_a = best_a.min(once(&mut a, &mut slots)); + } + } + (best_a, best_b) +} + +/// For every `surfacematerial` in every file named on the command line, hands +/// `bench` its name, a compiler for it (against the procedural texture; call +/// it once per variant wanted) and the shading points. +pub fn for_each_material(mut bench: impl FnMut(&str, &dyn Fn() -> Compiled, &[ShadeCtx])) { + let pts = points(); + for path in std::env::args().skip(1) { + let path = std::path::Path::new(&path); + let doc = Doc::open(path).expect("parse"); + for node in doc.by_category("surfacematerial") { + let compile = || compile(path, Some(&node.name), &Host::new(&procedural)).unwrap(); + bench(&node.name, &compile, &pts); + } + } +} diff --git a/crates/crust-mtlx/examples/mtlx_bench.rs b/crates/crust-mtlx/examples/mtlx_bench.rs index 03b19e43..d0220019 100644 --- a/crates/crust-mtlx/examples/mtlx_bench.rs +++ b/crates/crust-mtlx/examples/mtlx_bench.rs @@ -2,7 +2,7 @@ //! program run, per material, for the unoptimised and optimised programs. //! //! What a render spends on a MaterialX surface is dominated, after the -//! shade-once split, by one [`Program::eval`] per path vertex — so an +//! shade-once split, by one [`crust_mtlx::Program::eval`] per path vertex — so an //! interpreter change is measured here, in seconds, rather than through a //! ten-minute callgrind of the lion. Textures are a cheap procedural stand-in, //! which makes the interpreter's share larger than in a render: compare @@ -15,98 +15,29 @@ //! //! Every file given is benchmarked for every `surfacematerial` it holds. -use crust_mtlx::{Compiled, Doc, Host, Program, ShadeCtx, Texture, TextureRef, Val, compile}; -use glam::Vec3A; -use std::hint::black_box; -use std::sync::Arc; -use std::time::Instant; +#[path = "bench_common/mod.rs"] +mod bench_common; -struct Procedural; -impl Texture for Procedural { - fn eval(&self, u: f32, v: f32, width: f32) -> [f32; 4] { - [u.fract().abs(), v.fract().abs(), 0.5, 0.5 + width] - } -} - -fn procedural(_: &str, _: Option<&str>) -> Option { - Some(TextureRef(Arc::new(Procedural))) -} - -const POINTS: usize = 4096; -const REPEATS: usize = 15; - -fn points() -> Vec { - (0..POINTS) - .map(|i| { - let t = i as f32 / POINTS as f32; - ShadeCtx { - uv: (t * 2.0, (t * 17.0).fract()), - normal: Vec3A::new((t * 7.0).sin(), (t * 5.0).cos(), 1.0).normalize(), - tangent: Vec3A::X, - view: -Vec3A::new(0.2, (t * 3.0).sin(), 1.0).normalize(), - position: Vec3A::new(t, 1.0 - t, t * t), - uv_width: t * 0.01, - } - }) - .collect() -} - -/// Min-of-N nanoseconds per run for two evaluators, **interleaved**: each -/// repeat times both, alternating which goes first, so load or a frequency -/// change lands on both rather than on whichever phase it happened to hit -/// (the in-process version of `scripts/bench_ab.sh`). -fn time_ab( - pts: &[ShadeCtx], - mut a: impl FnMut(&ShadeCtx, &mut Vec), - mut b: impl FnMut(&ShadeCtx, &mut Vec), -) -> (f64, f64) { - let mut slots = Vec::new(); - let once = |run: &mut dyn FnMut(&ShadeCtx, &mut Vec), slots: &mut Vec| { - let t = Instant::now(); - for p in pts { - run(black_box(p), slots); - black_box(&*slots); - } - t.elapsed().as_nanos() as f64 / pts.len() as f64 - }; - let (mut best_a, mut best_b) = (f64::INFINITY, f64::INFINITY); - for rep in 0..REPEATS { - if rep % 2 == 0 { - best_a = best_a.min(once(&mut a, &mut slots)); - best_b = best_b.min(once(&mut b, &mut slots)); - } else { - best_b = best_b.min(once(&mut b, &mut slots)); - best_a = best_a.min(once(&mut a, &mut slots)); - } - } - (best_a, best_b) -} +use bench_common::{for_each_material, time_ab}; fn main() { - let pts = points(); println!( "{:<40} {:>6} {:>6} {:>10} {:>10}", "material", "ops", "opt", "ns/run", "ns/opt" ); - for path in std::env::args().skip(1) { - let path = std::path::Path::new(&path); - let doc = Doc::open(path).expect("parse"); - for node in doc.by_category("surfacematerial") { - let reference: Compiled = - compile(path, Some(&node.name), &Host::new(&procedural)).unwrap(); - let mut optimized: Compiled = - compile(path, Some(&node.name), &Host::new(&procedural)).unwrap(); - optimized.optimize(); - let (p, o): (&Program, &Program) = (&reference.program, &optimized.program); - let (t_ref, t_opt) = time_ab(&pts, |c, s| p.eval(c, s), |c, s| o.eval(c, s)); - println!( - "{:<40} {:>6} {:>6} {:>10.1} {:>10.1}", - node.name, - p.len(), - o.ops.len(), - t_ref, - t_opt - ); - } - } + for_each_material(|name, compile, pts| { + let reference = compile(); + let mut optimized = compile(); + optimized.optimize(); + let (p, o) = (&reference.program, &optimized.program); + let (t_ref, t_opt) = time_ab(pts, |c, s| p.eval(c, s), |c, s| o.eval(c, s)); + println!( + "{:<40} {:>6} {:>6} {:>10.1} {:>10.1}", + name, + p.len(), + o.ops.len(), + t_ref, + t_opt + ); + }); } diff --git a/crates/crust-render/examples/light_occlusion.rs b/crates/crust-render/examples/light_occlusion.rs index f493cbb2..6050c011 100644 --- a/crates/crust-render/examples/light_occlusion.rs +++ b/crates/crust-render/examples/light_occlusion.rs @@ -28,8 +28,8 @@ use crust_assets::FileAssets; use crust_core::{ - Light, LightSelection, MASK_SHADOW, Material, Ray, Renderer, ShadingPoint, UsdImportOptions, - Vec3A, World, + Light, LightSelection, MASK_SHADOW, Material, Ray, Renderer, ShadingPoint, TRACE_T_MIN, + UsdImportOptions, Vec3A, World, }; use std::path::PathBuf; @@ -116,10 +116,10 @@ fn transmissive(mat: &dyn Material, ray: &Ray, hit: &crust_core::HitRecord) -> b /// transmissive" covers only the ones it saw. fn walk(world: &World, from: Vec3A, dir: Vec3A, dist: f32) -> (usize, bool, f32, f32, bool) { let ray = Ray::new(from, dir).with_mask(MASK_SHADOW); - let (mut t0, mut n, mut glass) = (0.001f32, 0, true); + let (mut t0, mut n, mut glass) = (TRACE_T_MIN, 0, true); let (mut first, mut last) = (f32::NAN, f32::NAN); while n < MAX_CROSSINGS { - let Some(h) = world.intersect(&ray, t0, dist - 0.001) else { + let Some(h) = world.intersect(&ray, t0, dist - TRACE_T_MIN) else { break; }; if n == 0 { @@ -132,7 +132,7 @@ fn walk(world: &World, from: Vec3A, dir: Vec3A, dist: f32) -> (usize, bool, f32, // again by rounding. t0 = h.rec.t + 1e-4 * (1.0 + h.rec.t); } - let truncated = n == MAX_CROSSINGS && world.intersect(&ray, t0, dist - 0.001).is_some(); + let truncated = n == MAX_CROSSINGS && world.intersect(&ray, t0, dist - TRACE_T_MIN).is_some(); (n, glass, first, last, truncated) } @@ -221,7 +221,7 @@ fn main() { for i in 0..gw { let (u, v) = ((i as f32 + 0.5) / gw as f32, (j as f32 + 0.5) / gh as f32); let ray = r.camera.get_ray(u, v, [0.5, 0.5], 0.0); - let Some(hit) = world.intersect(&ray, 0.001, f32::INFINITY) else { + let Some(hit) = world.intersect(&ray, TRACE_T_MIN, f32::INFINITY) else { continue; }; receivers += 1; @@ -246,7 +246,11 @@ fn main() { let shadow = Ray::new(p, ls.direction).with_mask(MASK_SHADOW); if !carries { Outcome::BelowHorizon - } else if !world.occluded(&shadow, 0.001, ls.distance - 0.001) { + } else if !world.occluded( + &shadow, + TRACE_T_MIN, + ls.distance - TRACE_T_MIN, + ) { Outcome::Visible } else { // A light at infinity is walked to a far bound, diff --git a/crates/crust-rt/benches/traversal.rs b/crates/crust-rt/benches/traversal.rs index 0a7499da..b1c8db93 100644 --- a/crates/crust-rt/benches/traversal.rs +++ b/crates/crust-rt/benches/traversal.rs @@ -14,118 +14,13 @@ //! - `instances_*`: an instanced grid — traversal that recurses through //! transformed sub-scenes. +#[path = "../fixtures/mod.rs"] +mod fixtures; + use criterion::{Criterion, criterion_group, criterion_main}; -use crust_rt::{Geometry, Ray, Scene, SceneBuilder}; -use glam::{Affine3A, Vec3A}; +use crust_rt::{Scene, SceneBuilder}; +use fixtures::{T_MIN, instance_scene, ray_batch, sphere_grid_scene, triangle_scene}; use std::hint::black_box; -use std::sync::Arc; - -/// A UV sphere mesh with `2 * segs * rings` triangles. -fn uv_sphere(center: Vec3A, radius: f32, segs: usize, rings: usize) -> Geometry { - let mut vertices = Vec::with_capacity((segs + 1) * (rings + 1)); - for r in 0..=rings { - let v = r as f32 / rings as f32; - let phi = v * std::f32::consts::PI; - for s in 0..=segs { - let u = s as f32 / segs as f32; - let theta = u * std::f32::consts::TAU; - vertices.push( - center - + radius - * Vec3A::new(phi.sin() * theta.cos(), phi.cos(), phi.sin() * theta.sin()), - ); - } - } - let mut indices = Vec::with_capacity(2 * segs * rings); - let row = segs + 1; - for r in 0..rings { - for s in 0..segs { - let a = (r * row + s) as u32; - let b = (r * row + s + 1) as u32; - let c = ((r + 1) * row + s + 1) as u32; - let d = ((r + 1) * row + s) as u32; - indices.push([a, b, c]); - indices.push([a, c, d]); - } - } - Geometry::TriangleMesh { - vertices: vertices.iter().map(|v: &Vec3A| v.to_array()).collect(), - indices, - normals: None, - } -} - -/// Triangle-heavy scene: a 3×3×3 arrangement of subdivided spheres -/// (~86k triangles). -fn triangle_scene() -> Scene { - let mut b = SceneBuilder::new(); - for x in -1..=1 { - for y in -1..=1 { - for z in -1..=1 { - let c = Vec3A::new(x as f32, y as f32, z as f32) * 2.5; - b.attach(uv_sphere(c, 1.0, 40, 20)); - } - } - } - b.commit() -} - -/// Analytic spheres only — isolates node tests from triangle work. -fn sphere_grid_scene() -> Scene { - let mut b = SceneBuilder::new(); - for x in 0..12 { - for y in 0..12 { - for z in 0..12 { - b.attach(Geometry::Sphere { - center: Vec3A::new(x as f32, y as f32, z as f32) * 2.0 - Vec3A::splat(12.0), - radius: 0.6, - }); - } - } - } - b.commit() -} - -/// One mesh, instanced across a grid: two-level traversal. -fn instance_scene() -> Scene { - let mut inner = SceneBuilder::new(); - inner.attach(uv_sphere(Vec3A::ZERO, 1.0, 24, 12)); - let inner = Arc::new(inner.commit()); - - let mut b = SceneBuilder::new(); - for x in -2..=2 { - for y in -2..=2 { - for z in -2..=2 { - b.attach(Geometry::Instance { - scene: Arc::clone(&inner), - transform: Affine3A::from_translation( - glam::Vec3::new(x as f32, y as f32, z as f32) * 2.5, - ), - transform_end: None, - }); - } - } - } - b.commit() -} - -/// A deterministic fan of rays aimed through the scene's bounds — a mix of -/// hits and misses, and of coherent and divergent directions. Seeded by a -/// small LCG so the batch is identical run to run. -fn ray_batch(count: usize, extent: f32) -> Vec { - let mut state = 0x2545_F491u32; - let mut next = || { - state = state.wrapping_mul(1_664_525).wrapping_add(1_013_904_223); - (state >> 8) as f32 / (1u32 << 24) as f32 - }; - (0..count) - .map(|_| { - let origin = Vec3A::new(next() - 0.5, next() - 0.5, next() - 0.5) * (4.0 * extent); - let target = Vec3A::new(next() - 0.5, next() - 0.5, next() - 0.5) * extent; - Ray::new(origin, (target - origin).normalize()) - }) - .collect() -} const RAYS: usize = 4096; @@ -136,7 +31,7 @@ fn bench_scene(c: &mut Criterion, name: &str, scene: Scene, extent: f32) { b.iter(|| { let mut hits = 0usize; for r in &rays { - if scene.intersect(r, 0.001, f32::INFINITY).is_some() { + if scene.intersect(r, T_MIN, f32::INFINITY).is_some() { hits += 1; } } @@ -148,7 +43,7 @@ fn bench_scene(c: &mut Criterion, name: &str, scene: Scene, extent: f32) { b.iter(|| { let mut hits = 0usize; for r in &rays { - if scene.occluded(r, 0.001, f32::INFINITY) { + if scene.occluded(r, T_MIN, f32::INFINITY) { hits += 1; } } @@ -158,14 +53,19 @@ fn bench_scene(c: &mut Criterion, name: &str, scene: Scene, extent: f32) { } fn bench_traversal(c: &mut Criterion) { - bench_scene(c, "tri_spheres", triangle_scene(), 6.0); - bench_scene(c, "sphere_grid", sphere_grid_scene(), 14.0); - bench_scene(c, "instances", instance_scene(), 7.0); + bench_scene(c, "tri_spheres", triangle_scene(SceneBuilder::commit), 6.0); + bench_scene( + c, + "sphere_grid", + sphere_grid_scene(SceneBuilder::commit), + 14.0, + ); + bench_scene(c, "instances", instance_scene(SceneBuilder::commit), 7.0); } fn bench_build(c: &mut Criterion) { c.bench_function("build tri_spheres", |b| { - b.iter(|| black_box(triangle_scene().primitive_count())) + b.iter(|| black_box(triangle_scene(SceneBuilder::commit).primitive_count())) }); } diff --git a/crates/crust-rt/examples/ray_throughput.rs b/crates/crust-rt/examples/ray_throughput.rs index 54c3dc9e..667c0c11 100644 --- a/crates/crust-rt/examples/ray_throughput.rs +++ b/crates/crust-rt/examples/ray_throughput.rs @@ -26,111 +26,18 @@ //! bounces, and there are more of them so the touched working set is large //! too. Building takes tens of seconds and a few GiB. -use crust_rt::{CommitOptions, Geometry, PacketLayout, Ray, Scene, SceneBuilder}; +#[path = "../fixtures/mod.rs"] +mod fixtures; + +use crust_rt::{CommitOptions, Geometry, PacketLayout, Scene, SceneBuilder}; +use fixtures::{ + T_MIN, in_cube, instance_scene, ray_batch, sphere_grid_scene, triangle_scene, uv_sphere, +}; use glam::{Affine3A, Vec3A}; +use openqmc::pcg::Rng; use std::sync::Arc; use std::time::Instant; -fn uv_sphere(center: Vec3A, radius: f32, segs: usize, rings: usize) -> Geometry { - let mut vertices = Vec::with_capacity((segs + 1) * (rings + 1)); - for r in 0..=rings { - let phi = (r as f32 / rings as f32) * std::f32::consts::PI; - for s in 0..=segs { - let theta = (s as f32 / segs as f32) * std::f32::consts::TAU; - vertices.push( - center - + radius - * Vec3A::new(phi.sin() * theta.cos(), phi.cos(), phi.sin() * theta.sin()), - ); - } - } - let mut indices = Vec::with_capacity(2 * segs * rings); - let row = segs + 1; - for r in 0..rings { - for s in 0..segs { - let a = (r * row + s) as u32; - let b = (r * row + s + 1) as u32; - let c = ((r + 1) * row + s + 1) as u32; - let d = ((r + 1) * row + s) as u32; - indices.push([a, b, c]); - indices.push([a, c, d]); - } - } - Geometry::TriangleMesh { - vertices: vertices.iter().map(|v: &Vec3A| v.to_array()).collect(), - indices, - normals: None, - } -} - -fn triangle_scene() -> Scene { - let mut b = SceneBuilder::new(); - for x in -1..=1 { - for y in -1..=1 { - for z in -1..=1 { - b.attach(uv_sphere( - Vec3A::new(x as f32, y as f32, z as f32) * 2.5, - 1.0, - 40, - 20, - )); - } - } - } - commit(b) -} - -fn sphere_grid_scene() -> Scene { - let mut b = SceneBuilder::new(); - for x in 0..12 { - for y in 0..12 { - for z in 0..12 { - b.attach(Geometry::Sphere { - center: Vec3A::new(x as f32, y as f32, z as f32) * 2.0 - Vec3A::splat(12.0), - radius: 0.6, - }); - } - } - } - commit(b) -} - -fn instance_scene() -> Scene { - let mut inner = SceneBuilder::new(); - inner.attach(uv_sphere(Vec3A::ZERO, 1.0, 24, 12)); - let inner = Arc::new(commit(inner)); - let mut b = SceneBuilder::new(); - for x in -2..=2 { - for y in -2..=2 { - for z in -2..=2 { - b.attach(Geometry::Instance { - scene: Arc::clone(&inner), - transform: Affine3A::from_translation( - glam::Vec3::new(x as f32, y as f32, z as f32) * 2.5, - ), - transform_end: None, - }); - } - } - } - commit(b) -} - -/// A tiny deterministic generator for the large scenes (the same LCG as -/// [`ray_batch`], seeded differently). -struct Lcg(u32); - -impl Lcg { - fn next(&mut self) -> f32 { - self.0 = self.0.wrapping_mul(1_664_525).wrapping_add(1_013_904_223); - (self.0 >> 8) as f32 / (1u32 << 24) as f32 - } - - fn in_cube(&mut self, half: f32) -> Vec3A { - Vec3A::new(self.next() - 0.5, self.next() - 0.5, self.next() - 0.5) * (2.0 * half) - } -} - /// Half-extent of the large scenes' cube. const LARGE_HALF: f32 = 50.0; @@ -142,13 +49,13 @@ fn soup_scene(n: usize) -> Scene { let side = 2.0 * LARGE_HALF; let area = 8.0 * side * side / n as f32; let edge = (2.0 * area).sqrt(); - let mut rng = Lcg(0x9E37_79B9); + let mut rng = Rng::new(0x9E37_79B9); let mut vertices = Vec::with_capacity(3 * n); let mut indices = Vec::with_capacity(n); for i in 0..n { - let c = rng.in_cube(LARGE_HALF); + let c = in_cube(&mut rng, LARGE_HALF); for _ in 0..3 { - vertices.push(c + rng.in_cube(0.5 * edge)); + vertices.push(c + in_cube(&mut rng, 0.5 * edge)); } let b = 3 * i as u32; indices.push([b, b + 1, b + 2]); @@ -177,15 +84,15 @@ fn instance_field_scene(count: usize) -> Scene { let side = 2.0 * LARGE_HALF; // Cross-section pi r^2 per instance: mean free path V / (count pi r^2). let radius = (4.0 * side * side / (count as f32 * std::f32::consts::PI)).sqrt(); - let mut rng = Lcg(0x85EB_CA6B); + let mut rng = Rng::new(0x85EB_CA6B); let mut b = SceneBuilder::new(); for i in 0..count { - let axis = (rng.in_cube(1.0) + Vec3A::splat(1e-3)).normalize(); - let scale = radius * (0.5 + rng.next()); + let axis = (in_cube(&mut rng, 1.0) + Vec3A::splat(1e-3)).normalize(); + let scale = radius * (0.5 + rng.next_f32()); let transform = Affine3A::from_scale_rotation_translation( glam::Vec3::splat(scale), - glam::Quat::from_axis_angle(axis.into(), rng.next() * std::f32::consts::TAU), - rng.in_cube(LARGE_HALF).into(), + glam::Quat::from_axis_angle(axis.into(), rng.next_f32() * std::f32::consts::TAU), + in_cube(&mut rng, LARGE_HALF).into(), ); b.attach(Geometry::Instance { scene: Arc::clone(&protos[i % protos.len()]), @@ -196,21 +103,6 @@ fn instance_field_scene(count: usize) -> Scene { commit(b) } -fn ray_batch(count: usize, extent: f32) -> Vec { - let mut state = 0x2545_F491u32; - let mut next = || { - state = state.wrapping_mul(1_664_525).wrapping_add(1_013_904_223); - (state >> 8) as f32 / (1u32 << 24) as f32 - }; - (0..count) - .map(|_| { - let origin = Vec3A::new(next() - 0.5, next() - 0.5, next() - 0.5) * (4.0 * extent); - let target = Vec3A::new(next() - 0.5, next() - 0.5, next() - 0.5) * extent; - Ray::new(origin, (target - origin).normalize()) - }) - .collect() -} - const RAYS: usize = 4096; const REPEATS: usize = 40; /// The large scenes trace more rays, so the set of nodes one pass touches @@ -236,9 +128,9 @@ fn probe_with(name: &str, scene: &Scene, extent: f32, n_rays: usize, repeats: us let mut h = 0usize; for r in &rays { let hit = if closest { - scene.intersect(r, 0.001, f32::INFINITY).is_some() + scene.intersect(r, T_MIN, f32::INFINITY).is_some() } else { - scene.occluded(r, 0.001, f32::INFINITY) + scene.occluded(r, T_MIN, f32::INFINITY) }; if hit { h += 1; @@ -284,12 +176,8 @@ fn commit(b: SceneBuilder) -> Scene { } fn main() { - let tri = triangle_scene(); - println!("tri_spheres: {} triangles", tri.primitive_count()); - probe("tri_spheres", &tri, 6.0); - probe("sphere_grid", &sphere_grid_scene(), 14.0); - probe("instances", &instance_scene(), 7.0); - + // `--layout` is read before any scene is built, so the default scenes + // commit with it too. let mut args = std::env::args().skip(1).peekable(); if args.peek().map(String::as_str) == Some("--layout") { args.next(); @@ -301,6 +189,12 @@ fn main() { }; LAYOUT.set(layout).expect("set once"); } + + let tri = triangle_scene(commit); + println!("tri_spheres: {} triangles", tri.primitive_count()); + probe("tri_spheres", &tri, 6.0); + probe("sphere_grid", &sphere_grid_scene(commit), 14.0); + probe("instances", &instance_scene(commit), 7.0); if args.next().as_deref() == Some("--large") { let mtris: f64 = args.next().and_then(|a| a.parse().ok()).unwrap_or(8.0); let n = (mtris * 1e6) as usize; diff --git a/crates/crust-rt/examples/traversal_probe.rs b/crates/crust-rt/examples/traversal_probe.rs index b3844edd..e2eba708 100644 --- a/crates/crust-rt/examples/traversal_probe.rs +++ b/crates/crust-rt/examples/traversal_probe.rs @@ -12,42 +12,15 @@ //! Trust the counts from this build, not its timings: the counters are //! global atomics and contend across threads. +#[path = "../fixtures/mod.rs"] +mod fixtures; + use crust_rt::{Geometry, Ray, SceneBuilder}; use glam::Vec3A; -/// A UV sphere, as a stand-in for real mesh geometry at a chosen size. +/// A unit UV sphere at the origin. fn uv_sphere(segs: usize, rings: usize) -> Geometry { - let mut vertices = Vec::new(); - for r in 0..=rings { - let phi = (r as f32 / rings as f32) * std::f32::consts::PI; - for s in 0..=segs { - let th = (s as f32 / segs as f32) * std::f32::consts::TAU; - vertices.push(Vec3A::new( - phi.sin() * th.cos(), - phi.cos(), - phi.sin() * th.sin(), - )); - } - } - let row = segs + 1; - let mut indices = Vec::new(); - for r in 0..rings { - for s in 0..segs { - let (a, b, c, d) = ( - (r * row + s) as u32, - (r * row + s + 1) as u32, - ((r + 1) * row + s + 1) as u32, - ((r + 1) * row + s) as u32, - ); - indices.push([a, b, c]); - indices.push([a, c, d]); - } - } - Geometry::TriangleMesh { - vertices: vertices.iter().map(|v: &Vec3A| v.to_array()).collect(), - indices, - normals: None, - } + fixtures::uv_sphere(Vec3A::ZERO, 1.0, segs, rings) } fn probe(label: &str, geom: Geometry) { @@ -70,7 +43,10 @@ fn probe(label: &str, geom: Geometry) { // node test and hide the real traversal depth. let dir = Vec3A::new(0.28 * x as f32 / N as f32, 0.28 * y as f32 / N as f32, 1.0); let ray = Ray::new(Vec3A::new(0.0, 0.0, -4.0), dir); - if scene.intersect(&ray, 0.001, f32::INFINITY).is_some() { + if scene + .intersect(&ray, fixtures::T_MIN, f32::INFINITY) + .is_some() + { hits += 1; } } @@ -181,7 +157,10 @@ fn probe_instanced(label: &str, copies: usize, segs: usize, rings: usize) { 1.0, ); let ray = Ray::new(Vec3A::new(0.0, 0.0, -dist), dir); - if scene.intersect(&ray, 0.001, f32::INFINITY).is_some() { + if scene + .intersect(&ray, fixtures::T_MIN, f32::INFINITY) + .is_some() + { hits += 1; } } @@ -267,7 +246,10 @@ fn probe_nested(label: &str, groups: usize, per_group: usize, segs: usize, rings 1.0, ); let ray = Ray::new(Vec3A::new(0.0, 0.0, -dist), dir); - if scene.intersect(&ray, 0.001, f32::INFINITY).is_some() { + if scene + .intersect(&ray, fixtures::T_MIN, f32::INFINITY) + .is_some() + { hits += 1; } } diff --git a/crates/crust-rt/fixtures/mod.rs b/crates/crust-rt/fixtures/mod.rs new file mode 100644 index 00000000..c70dbaa0 --- /dev/null +++ b/crates/crust-rt/fixtures/mod.rs @@ -0,0 +1,129 @@ +//! Scene and ray fixtures shared by crust-rt's throughput probes and its +//! criterion bench (`examples/ray_throughput.rs`, `examples/traversal_probe.rs`, +//! `benches/traversal.rs`), included by each with `#[path]` so they time the +//! same geometry and the same rays. +//! +//! Randomness comes from `openqmc::pcg::Rng` (a dev-dependency), the +//! repository's one source of seeded draws outside the renderer's samplers. + +#![allow(dead_code)] // each including target uses a different subset + +use crust_rt::{Geometry, Ray, Scene, SceneBuilder}; +use glam::{Affine3A, Vec3A}; +use openqmc::pcg::Rng; +use std::sync::Arc; + +/// The near bound every query here asks for — the renderer's +/// `crust_core::TRACE_T_MIN`, so the kernel is timed with the constant it is +/// specialised for. +pub const T_MIN: f32 = 0.001; + +/// A UV sphere mesh with `2 * segs * rings` triangles — a stand-in for real +/// mesh geometry at a chosen size. +pub fn uv_sphere(center: Vec3A, radius: f32, segs: usize, rings: usize) -> Geometry { + let mut vertices = Vec::with_capacity((segs + 1) * (rings + 1)); + for r in 0..=rings { + let phi = (r as f32 / rings as f32) * std::f32::consts::PI; + for s in 0..=segs { + let theta = (s as f32 / segs as f32) * std::f32::consts::TAU; + vertices.push( + center + + radius + * Vec3A::new(phi.sin() * theta.cos(), phi.cos(), phi.sin() * theta.sin()), + ); + } + } + let mut indices = Vec::with_capacity(2 * segs * rings); + let row = segs + 1; + for r in 0..rings { + for s in 0..segs { + let a = (r * row + s) as u32; + let b = (r * row + s + 1) as u32; + let c = ((r + 1) * row + s + 1) as u32; + let d = ((r + 1) * row + s) as u32; + indices.push([a, b, c]); + indices.push([a, c, d]); + } + } + Geometry::TriangleMesh { + vertices: vertices.iter().map(|v: &Vec3A| v.to_array()).collect(), + indices, + normals: None, + } +} + +/// Triangle-heavy: a 3×3×3 arrangement of subdivided spheres (~86k +/// triangles), several per leaf, which is what the 4-wide leaf intersector +/// targets. Committed by `commit`, so a caller can choose the options. +pub fn triangle_scene(commit: impl Fn(SceneBuilder) -> Scene) -> Scene { + let mut b = SceneBuilder::new(); + for x in -1..=1 { + for y in -1..=1 { + for z in -1..=1 { + let c = Vec3A::new(x as f32, y as f32, z as f32) * 2.5; + b.attach(uv_sphere(c, 1.0, 40, 20)); + } + } + } + commit(b) +} + +/// Analytic spheres only — isolates the node slab test and traversal +/// ordering from triangle work. +pub fn sphere_grid_scene(commit: impl Fn(SceneBuilder) -> Scene) -> Scene { + let mut b = SceneBuilder::new(); + for x in 0..12 { + for y in 0..12 { + for z in 0..12 { + b.attach(Geometry::Sphere { + center: Vec3A::new(x as f32, y as f32, z as f32) * 2.0 - Vec3A::splat(12.0), + radius: 0.6, + }); + } + } + } + commit(b) +} + +/// One mesh instanced across a 5×5×5 grid: two-level traversal through +/// transformed sub-scenes. +pub fn instance_scene(commit: impl Fn(SceneBuilder) -> Scene) -> Scene { + let mut inner = SceneBuilder::new(); + inner.attach(uv_sphere(Vec3A::ZERO, 1.0, 24, 12)); + let inner = Arc::new(commit(inner)); + let mut b = SceneBuilder::new(); + for x in -2..=2 { + for y in -2..=2 { + for z in -2..=2 { + b.attach(Geometry::Instance { + scene: Arc::clone(&inner), + transform: Affine3A::from_translation( + glam::Vec3::new(x as f32, y as f32, z as f32) * 2.5, + ), + transform_end: None, + }); + } + } + } + commit(b) +} + +/// A uniform point in the cube of half-extent `half` about the origin. +pub fn in_cube(rng: &mut Rng, half: f32) -> Vec3A { + let [x, y] = rng.next_2d(); + Vec3A::new(x - 0.5, y - 0.5, rng.next_f32() - 0.5) * (2.0 * half) +} + +/// A deterministic batch of rays aimed through the scene's bounds — a mix of +/// hits and misses, and of coherent and divergent directions: origins in a +/// cube of half-extent `2 · extent`, targets in one of `extent / 2`. +pub fn ray_batch(count: usize, extent: f32) -> Vec { + let mut rng = Rng::new(0x2545_F491); + (0..count) + .map(|_| { + let origin = in_cube(&mut rng, 2.0 * extent); + let target = in_cube(&mut rng, 0.5 * extent); + Ray::new(origin, (target - origin).normalize()) + }) + .collect() +} diff --git a/crates/utils/src/common.rs b/crates/utils/src/common.rs index 11030337..89d8a2ba 100644 --- a/crates/utils/src/common.rs +++ b/crates/utils/src/common.rs @@ -1,6 +1,13 @@ use glam::Vec3A; use std::f32::consts::PI; +/// `e^v` per component — Beer–Lambert transmittance `exp3(-σ·t)` and its +/// kin, wherever a colour is exponentiated. +#[inline] +pub fn exp3(v: Vec3A) -> Vec3A { + Vec3A::new(v.x.exp(), v.y.exp(), v.z.exp()) +} + pub fn degrees_to_radians(degrees: f32) -> f32 { degrees * PI / 180.0 } diff --git a/crates/utils/src/lib.rs b/crates/utils/src/lib.rs index c1f844c3..8007efc8 100644 --- a/crates/utils/src/lib.rs +++ b/crates/utils/src/lib.rs @@ -3,7 +3,7 @@ mod common; pub use common::Lerp; pub use common::{ - Luma, align_to_normal, concentric_disk, cosine_hemisphere, degrees_to_radians, luminance, + Luma, align_to_normal, concentric_disk, cosine_hemisphere, degrees_to_radians, exp3, luminance, uniform_ball, uniform_sphere, }; pub use common::{balance_heuristic, power_heuristic}; From 71db2d63e774b0efad4c4625b1d180f99883784e Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 5 Oct 2026 07:42:40 +0000 Subject: [PATCH 3/5] Narrow public APIs, retire CRUST_BVH_PACKET_SAH, fix error-prone APIs - LightList: drop find_by_geom, pick_at, find_by_geom_at, iter_at and infinite_seen_by, which only tests used; tests read the index forms. - Retire CRUST_BVH_PACKET_SAH (its A/B is settled in the design record): packet-sized leaves are the only rule for all-triangle ranges, the per-primitive cost stays for ranges holding anything else. Removed from Config, CommitOptions and the builder; docs, specs and site updated. - crust-core: subsurface and commit_options are crate-private. - crust-mtlx: modules are private, every item has one path at the root (the nodedef tables and parse_literal/arity_of the tests use included). - crust-assets: tiled is private; make_tx/make_tx_atomic/TxFormat/MadeTx re-exported at the root, and the dead accessors that exposed removed. - RenderSettings::new(seven positional numbers) becomes Default plus with_resolution / with_max_depth / with_adaptive_sampling builders. - HitRecord carries uv: Option<(f32, f32)> instead of uv plus has_uv. - RayStats::merge destructures, so a new counter cannot be forgotten. - Guiding's quadtree descent draws from the K_GUIDE domain's rng() instead of a hand-rolled PCG32. cornellbox_guided changes by noise only: relmse 3.1e-2 / 2.1e-2 / 7.6e-3 at 16/64/256 spp against the old stream, equally far from an unguided reference. Every other sample is bit-identical. - Doc comments moved to their items (sample_bounce_direction in path.rs, triangulate in mesh.rs); tex_probe's gamma-2.2 lines say so. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_019nwge6NCTPhuk1VPuZucRF --- crates/crust-assets/src/lib.rs | 4 +- crates/crust-assets/src/tiled/cache.rs | 31 ------- crates/crust-assets/src/tiled/mod.rs | 5 +- crates/crust-assets/tests/auto_tx.rs | 2 +- crates/crust-assets/tests/wide_gamut.rs | 2 +- crates/crust-core/benches/integrator.rs | 30 +++---- crates/crust-core/src/config.rs | 6 -- crates/crust-core/src/guiding/dtree.rs | 41 +++------ crates/crust-core/src/guiding/field.rs | 11 +-- crates/crust-core/src/guiding/sdtree.rs | 2 +- crates/crust-core/src/hittable.rs | 15 ++-- crates/crust-core/src/lib.rs | 12 ++- crates/crust-core/src/light/list.rs | 72 ++++------------ crates/crust-core/src/light/tests.rs | 24 +++--- crates/crust-core/src/material/closure/mod.rs | 2 +- crates/crust-core/src/material/material.rs | 2 +- crates/crust-core/src/material/materialx.rs | 2 +- .../src/material/preview_surface.rs | 5 +- crates/crust-core/src/rt_world.rs | 11 ++- .../crust-core/src/scene/usd_import/mesh.rs | 9 +- .../src/scene/usd_import/settings.rs | 42 ++++------ crates/crust-core/src/stats.rs | 84 +++++++++++++------ crates/crust-core/src/tracer/path.rs | 31 +++---- crates/crust-core/src/tracer/settings.rs | 57 ++++++++----- crates/crust-core/src/world.rs | 6 +- crates/crust-core/tests/aovs.rs | 8 +- crates/crust-core/tests/camera_buffer_ray.rs | 3 +- crates/crust-core/tests/guiding.rs | 25 ++++-- crates/crust-core/tests/guiding_field.rs | 23 +++-- crates/crust-core/tests/hair.rs | 3 +- crates/crust-core/tests/learned_selection.rs | 39 ++++----- crates/crust-core/tests/lights.rs | 14 ++-- crates/crust-core/tests/lpe.rs | 13 ++- crates/crust-core/tests/mtlx_surfaces.rs | 3 +- crates/crust-core/tests/profile.rs | 6 +- crates/crust-core/tests/render_smoke.rs | 83 ++++++++++++++---- crates/crust-core/tests/resolve.rs | 3 +- crates/crust-core/tests/stats.rs | 6 +- crates/crust-core/tests/usd_inline.rs | 14 ++-- crates/crust-core/tests/usd_scene.rs | 31 +++---- crates/crust-core/tests/world_material.rs | 18 ++-- crates/crust-mtlx/src/eval.rs | 8 +- crates/crust-mtlx/src/lib.rs | 31 ++++--- crates/crust-mtlx/src/parse.rs | 2 +- crates/crust-mtlx/tests/graph.rs | 2 +- crates/crust-mtlx/tests/nodedefs.rs | 6 +- crates/crust-render/examples/maketx.rs | 7 +- crates/crust-render/examples/mtlx_shade.rs | 3 +- crates/crust-render/examples/tex_probe.rs | 4 +- crates/crust-rt/examples/ray_throughput.rs | 1 - crates/crust-rt/src/bvh/build.rs | 37 ++++---- crates/crust-rt/src/bvh/lane_width.rs | 4 +- crates/crust-rt/src/bvh/mod.rs | 4 +- crates/crust-rt/src/bvh/tests.rs | 43 ++++------ crates/crust-rt/src/scene.rs | 13 +-- crates/crust-rt/tests/kernel.rs | 5 +- docs/architecture.md | 3 +- docs/light_sampling.md | 7 +- docs/simd.md | 2 +- openspec/specs/cli/design.md | 5 -- openspec/specs/cli/spec.md | 9 +- openspec/specs/intersection-kernel/design.md | 7 +- openspec/specs/intersection-kernel/spec.md | 18 ++-- openspec/specs/lighting/design.md | 8 +- openspec/specs/rendering/design.md | 9 +- .../docs/reference/environment-variables.md | 11 --- 66 files changed, 515 insertions(+), 534 deletions(-) diff --git a/crates/crust-assets/src/lib.rs b/crates/crust-assets/src/lib.rs index 663d3fd4..63bd787b 100644 --- a/crates/crust-assets/src/lib.rs +++ b/crates/crust-assets/src/lib.rs @@ -32,7 +32,7 @@ mod mip_filter; mod ptex_stream; mod ptex_texture; mod texture_cache; -pub mod tiled; +mod tiled; mod uv_texture; pub use environment::{load_exr_environment, load_image_environment, read_exr_rgb, read_rgb_image}; @@ -48,6 +48,8 @@ pub use ptex_stream::{ pub use ptex_texture::{ DEFAULT_MAX_LOG2, PtexColor, max_log2_from_env, max_log2_from_env_opt, read_channel, }; +/// Offline `.tx` conversion: what `maketx` and `--auto-tx` write. +pub use tiled::{MadeTx, TxFormat, make_tx, make_tx_atomic}; pub use uv_texture::{DEFAULT_MAX_EDGE, UvTexture}; /// The files a texture path names: every `` / `` tile on diff --git a/crates/crust-assets/src/tiled/cache.rs b/crates/crust-assets/src/tiled/cache.rs index 273e6bb0..4cbe2343 100644 --- a/crates/crust-assets/src/tiled/cache.rs +++ b/crates/crust-assets/src/tiled/cache.rs @@ -123,9 +123,6 @@ fn thread_stripe() -> usize { }) } -/// Default cache budget, matching OIIO's own 1 GB. -pub const DEFAULT_BUDGET_BYTES: u64 = mib_to_bytes(crust_core::config::DEFAULT_CACHE_MB as u64); - /// Which tile, of which level, of which file. /// /// The file is an interned index, not a path: a key is compared and hashed on @@ -210,10 +207,6 @@ impl TileData { TileKind::Half => self.bytes.len() / 2, } } - - pub fn is_empty(&self) -> bool { - self.len() == 0 - } } #[cfg(test)] @@ -364,21 +357,6 @@ pub struct CacheCounters { pub total_bytes: u64, } -impl CacheCounters { - pub fn is_empty(&self) -> bool { - self.micro_hits == 0 && self.hits == 0 && self.misses == 0 - } - - /// Fraction of lookups answered without touching the disk. - pub fn hit_rate(&self) -> f64 { - let total = self.micro_hits + self.hits + self.misses; - if total == 0 { - return 0.0; - } - (self.micro_hits + self.hits) as f64 / total as f64 - } -} - /// One file's geometry plus the decoders that read it. struct FileSlot { file: TiledFile, @@ -440,10 +418,6 @@ impl TileCache { mib_to_bytes(config.tex_cache_mb.get()) } - pub fn budget(&self) -> u64 { - self.budget - } - pub fn resident(&self) -> u64 { self.resident.load(Ordering::Relaxed) } @@ -460,11 +434,6 @@ impl TileCache { Some(id) } - pub fn file(&self, id: u32) -> Option { - let files = lock(&self.files)?; - files.get(id as usize).map(|s| s.file.clone()) - } - pub fn counters(&self) -> CacheCounters { let s = &self.stats; // Distinct tiles ever decoded, across every file. diff --git a/crates/crust-assets/src/tiled/mod.rs b/crates/crust-assets/src/tiled/mod.rs index 27df0b6c..7b319a40 100644 --- a/crates/crust-assets/src/tiled/mod.rs +++ b/crates/crust-assets/src/tiled/mod.rs @@ -45,10 +45,7 @@ mod read; mod stream; mod write; -pub use cache::{ - CacheCounters, DEFAULT_BUDGET_BYTES, StripedCounter, Tile, TileCache, TileData, TileId, - with_tile, -}; +pub use cache::{StripedCounter, TileCache, TileData}; pub(crate) use exr_read::exr_mip_space; pub use exr_write::write_tx_exr; pub use make::{MadeTx, TxFormat, is_ptex, make_tx, make_tx_atomic, tx_is_stale, tx_sibling}; diff --git a/crates/crust-assets/tests/auto_tx.rs b/crates/crust-assets/tests/auto_tx.rs index 9f1498b5..3a30e527 100644 --- a/crates/crust-assets/tests/auto_tx.rs +++ b/crates/crust-assets/tests/auto_tx.rs @@ -7,7 +7,7 @@ //! colour a lookup returns says which file answered. use crust_assets::FileAssets; -use crust_assets::tiled::{TxFormat, make_tx}; +use crust_assets::{TxFormat, make_tx}; use crust_core::{AssetLoader, ColorSpace}; use std::path::{Path, PathBuf}; diff --git a/crates/crust-assets/tests/wide_gamut.rs b/crates/crust-assets/tests/wide_gamut.rs index 1ded1624..8d6ecffb 100644 --- a/crates/crust-assets/tests/wide_gamut.rs +++ b/crates/crust-assets/tests/wide_gamut.rs @@ -4,7 +4,7 @@ //! an environment map — landing on the same numbers. use crust_assets::FileAssets; -use crust_assets::tiled::{TxFormat, make_tx}; +use crust_assets::{TxFormat, make_tx}; use crust_core::color::{Space, convert, working_space}; use crust_core::{AssetLoader, ColorSpace, Vec3A}; use std::path::{Path, PathBuf}; diff --git a/crates/crust-core/benches/integrator.rs b/crates/crust-core/benches/integrator.rs index 0ef022ba..5210b464 100644 --- a/crates/crust-core/benches/integrator.rs +++ b/crates/crust-core/benches/integrator.rs @@ -39,15 +39,11 @@ fn bench_simple_world(c: &mut Criterion) { aperture, dist_to_focus, ); - let render_settings = RenderSettings::new( - 10, - 20, - IMAGE_WIDTH, - IMAGE_HEIGHT, - MIN_SAMPLES, - VARIANCE_THRESHOLD, - 0, - ); + let render_settings = RenderSettings::default() + .with_resolution(IMAGE_WIDTH, IMAGE_HEIGHT) + .with_samples_per_pixel(10) + .with_max_depth(20) + .with_adaptive_sampling(MIN_SAMPLES, VARIANCE_THRESHOLD); let renderer = Renderer::new(cam, world, lights, render_settings); let _ = renderer.render(); }) @@ -73,16 +69,12 @@ fn bench_simple_world_guided(c: &mut Criterion) { aperture, dist_to_focus, ); - let render_settings = RenderSettings::new( - 10, - 20, - IMAGE_WIDTH, - IMAGE_HEIGHT, - MIN_SAMPLES, - VARIANCE_THRESHOLD, - 0, - ) - .with_guiding(true, 2, 0.5); + let render_settings = RenderSettings::default() + .with_resolution(IMAGE_WIDTH, IMAGE_HEIGHT) + .with_samples_per_pixel(10) + .with_max_depth(20) + .with_adaptive_sampling(MIN_SAMPLES, VARIANCE_THRESHOLD) + .with_guiding(true, 2, 0.5); let renderer = Renderer::new(cam, world, lights, render_settings); let _ = renderer.render(); }) diff --git a/crates/crust-core/src/config.rs b/crates/crust-core/src/config.rs index d23b1050..4194ffb0 100644 --- a/crates/crust-core/src/config.rs +++ b/crates/crust-core/src/config.rs @@ -167,10 +167,6 @@ pub struct Config { /// `CRUST_TRI_PACKETS`: the kernel's triangle packet layout /// (`gathered` | `indexed` | `auto`; bit-identical either way). pub tri_packets: TriPackets, - /// `CRUST_BVH_PACKET_SAH`: size all-triangle BVH leaves by SIMD packet - /// tests (`false`: the per-triangle leaf cost before the switch; the - /// trees differ in shape, so renders can differ on exact-tie hits). - pub bvh_packet_sah: bool, /// `CRUST_MTLX_OPT`: fold, hoist and prune MaterialX programs (`false`: /// run them as compiled; bit-identical). pub mtlx_opt: bool, @@ -240,7 +236,6 @@ impl Default for Config { adaptive_per_face: true, adaptive_frustum: true, tri_packets: TriPackets::Auto, - bvh_packet_sah: true, mtlx_opt: true, shader_jit: true, ray_cones: true, @@ -286,7 +281,6 @@ impl Config { d.tri_packets, "`gathered`, `indexed` or `auto`", ), - bvh_packet_sah: flag("CRUST_BVH_PACKET_SAH", d.bvh_packet_sah), mtlx_opt: flag("CRUST_MTLX_OPT", d.mtlx_opt), shader_jit: flag("CRUST_SHADER_JIT", d.shader_jit), ray_cones: flag("CRUST_RAY_CONES", d.ray_cones), diff --git a/crates/crust-core/src/guiding/dtree.rs b/crates/crust-core/src/guiding/dtree.rs index 34ca3dcd..c2d5ff6b 100644 --- a/crates/crust-core/src/guiding/dtree.rs +++ b/crates/crust-core/src/guiding/dtree.rs @@ -14,18 +14,6 @@ use std::f32::consts::{PI, TAU}; const NO_CHILD: u32 = u32::MAX; const ONE_MINUS_EPS: f32 = 1.0 - f32::EPSILON; -/// PCG32 step producing a uniform f32 in [0, 1). -#[inline] -fn pcg_f32(state: &mut u64) -> f32 { - *state = state - .wrapping_mul(6364136223846793005) - .wrapping_add(1442695040888963407); - let xorshifted = (((*state >> 18) ^ *state) >> 27) as u32; - let rot = (*state >> 59) as u32; - let bits = xorshifted.rotate_right(rot); - (bits >> 8) as f32 * (1.0 / (1u32 << 24) as f32) -} - /// Cylindrical equal-area map from a unit direction to the canonical square. pub fn dir_to_canonical(d: Vec3A) -> [f32; 2] { let d = d.normalize(); @@ -153,20 +141,17 @@ impl DTree { /// Draw a canonical position proportional to the stored flux, returning /// it with its solid-angle pdf. `None` if the tree holds no flux yet. /// - /// Takes a single 2D `seed` (one QMC domain draw); it seeds a PCG stream - /// that supplies fresh uniforms per tree level. Rescaling a single 2D - /// sample down the tree (the textbook trick) loses entropy on sharp - /// distributions until deep cells are no longer sampled uniformly, and - /// drawing per-level from the QMC sampler burns through its dimension - /// window; the hashed stream avoids both while keeping the sampler's - /// dimension usage fixed. + /// Takes an incidental stream (the vertex domain's `rng()`) that supplies + /// fresh uniforms per tree level. Rescaling a single 2D sample down the + /// tree (the textbook trick) loses entropy on sharp distributions until + /// deep cells are no longer sampled uniformly, and drawing per-level from + /// the QMC sampler burns through its dimension window; the stream avoids + /// both while keeping the sampler's dimension usage fixed. #[must_use] - pub fn sample(&self, seed: [f32; 2]) -> Option<([f32; 2], f32)> { + pub fn sample(&self, rng: &mut openqmc::pcg::Rng) -> Option<([f32; 2], f32)> { if self.total_flux() <= 0.0 { return None; } - let mut rng_state: u64 = - ((seed[0].to_bits() as u64) << 32 | seed[1].to_bits() as u64) ^ 0x9E37_79B9_7F4A_7C15; let mut node = 0usize; let mut base = [0.0f32; 2]; let mut scale = 1.0f32; @@ -178,7 +163,7 @@ impl DTree { // No information below this node: uniform within its domain. break; } - let u = [pcg_f32(&mut rng_state), pcg_f32(&mut rng_state)]; + let u = rng.next_2d(); // Pick the column proportional to column flux. let p_left = (n.sums[0] + n.sums[2]) / total; @@ -211,7 +196,7 @@ impl DTree { node = c as usize; } // Uniform position within the reached cell. - let u = [pcg_f32(&mut rng_state), pcg_f32(&mut rng_state)]; + let u = rng.next_2d(); let p = [ (base[0] + u[0] * scale).clamp(0.0, ONE_MINUS_EPS), (base[1] + u[1] * scale).clamp(0.0, ONE_MINUS_EPS), @@ -372,7 +357,7 @@ mod tests { let tree = trained_tree(); let mut s = Rng::new(0xC0FFEE); for _ in 0..1000 { - let (p, pdf) = tree.sample(s.next_2d()).expect("trained tree samples"); + let (p, pdf) = tree.sample(&mut s).expect("trained tree samples"); let lookup = tree.pdf(p); assert!( (pdf - lookup).abs() < 1e-3 * (1.0 + pdf), @@ -393,7 +378,7 @@ mod tests { } let tree = tree.refine(0.01, 20); for _ in 0..1000 { - let (p, _) = tree.sample(s.next_2d()).unwrap(); + let (p, _) = tree.sample(&mut s).unwrap(); let d = canonical_to_dir(p); assert!(d.z > -1e-3, "sampled below the trained hemisphere: {d}"); } @@ -428,7 +413,7 @@ mod tests { let n = 400_000; let mut counts = [0u64; 8]; for _ in 0..n { - let (p, _) = tree.sample(s.next_2d()).unwrap(); + let (p, _) = tree.sample(&mut s).unwrap(); let d = canonical_to_dir(p); let oct = (d.x >= 0.0) as usize + 2 * ((d.y >= 0.0) as usize) + 4 * ((d.z >= 0.0) as usize); @@ -466,7 +451,7 @@ mod tests { fn untrained_tree_declines_to_sample() { let tree = DTree::new(); let mut s = Rng::new(0xC0FFEE); - assert!(tree.sample(s.next_2d()).is_none()); + assert!(tree.sample(&mut s).is_none()); assert_eq!(tree.pdf([0.3, 0.7]), 0.0); } } diff --git a/crates/crust-core/src/guiding/field.rs b/crates/crust-core/src/guiding/field.rs index 9a63d4ba..73520bac 100644 --- a/crates/crust-core/src/guiding/field.rs +++ b/crates/crust-core/src/guiding/field.rs @@ -98,10 +98,11 @@ impl GuidingField { /// Draw a world-space direction from the local guiding distribution with /// its solid-angle pdf. `None` while the local distribution is untrained. - /// `seed` is a single 2D QMC domain draw (see [`crate::guiding::DTree::sample`]). + /// `rng` is the vertex domain's incidental stream; the quadtree descent + /// draws a fresh pair per level from it. #[must_use] - pub fn sample(&self, pos: Vec3A, seed: [f32; 2]) -> Option<(Vec3A, f32)> { - let (canonical, pdf) = self.tree.dtree_at(pos).sample(seed)?; + pub fn sample(&self, pos: Vec3A, rng: &mut openqmc::pcg::Rng) -> Option<(Vec3A, f32)> { + let (canonical, pdf) = self.tree.dtree_at(pos).sample(rng)?; Some((canonical_to_dir(canonical), pdf)) } @@ -143,7 +144,7 @@ mod tests { let mut field = GuidingField::new(bounds, GuidingConfig::default()); let mut s = openqmc::pcg::Rng::new(1); assert!(!field.trained_at(Vec3A::splat(0.5))); - assert!(field.sample(Vec3A::splat(0.5), s.next_2d()).is_none()); + assert!(field.sample(Vec3A::splat(0.5), &mut s).is_none()); let samples: Vec = (0..1000) .map(|i| SampleData { @@ -155,7 +156,7 @@ mod tests { field.update(&samples, 1); assert!(field.trained_at(Vec3A::splat(0.5))); - let (dir, pdf) = field.sample(Vec3A::splat(0.5), s.next_2d()).unwrap(); + let (dir, pdf) = field.sample(Vec3A::splat(0.5), &mut s).unwrap(); assert!(pdf > 0.0); assert!(dir.z > 0.0, "trained on +z but sampled {dir}"); assert!(field.pdf(Vec3A::splat(0.5), dir) > 0.0); diff --git a/crates/crust-core/src/guiding/sdtree.rs b/crates/crust-core/src/guiding/sdtree.rs index 638051ae..6e19d222 100644 --- a/crates/crust-core/src/guiding/sdtree.rs +++ b/crates/crust-core/src/guiding/sdtree.rs @@ -220,7 +220,7 @@ mod tests { for p in [Vec3A::new(0.1, 0.5, 0.5), Vec3A::new(0.9, 0.5, 0.5)] { let dtree = tree.dtree_at(p); assert!(dtree.total_flux() > 0.0, "child at {p} lost its flux"); - let (c, _) = dtree.sample(s.next_2d()).unwrap(); + let (c, _) = dtree.sample(&mut s).unwrap(); let d = super::super::dtree::canonical_to_dir(c); assert!(d.z > 0.0, "child at {p} samples away from the light: {d}"); } diff --git a/crates/crust-core/src/hittable.rs b/crates/crust-core/src/hittable.rs index 43c52875..ac337031 100644 --- a/crates/crust-core/src/hittable.rs +++ b/crates/crust-core/src/hittable.rs @@ -29,9 +29,10 @@ pub struct HitRecord { /// set addresses its tiles by the integer part, so clamping here would /// collapse fourteen 4K tiles onto one. /// - /// Left at `(0, 0)` — with `has_uv` false — for geometry carrying no `st` - /// primvar, or whose material asks for no UV texture. - pub uv: (f32, f32), + /// `None` for geometry carrying no `st` primvar, or whose material asks + /// for no UV texture — never a sentinel value, since the origin of the + /// chart is a perfectly ordinary texel. + pub uv: Option<(f32, f32)>, /// World-space surface tangent along increasing `u`, already /// orthogonalised against `normal` and unit length. The frame a /// tangent-space normal map is expressed in; the bitangent is @@ -42,11 +43,6 @@ pub struct HitRecord { /// The importer can only build one for *baked* (single-placement) /// geometry — see [`crate::UvMap::tangents`]. pub tangent: Vec3A, - /// Whether `uv` carries a real texture coordinate. - /// - /// Distinct from `uv != (0, 0)`: the origin of the chart is a perfectly - /// ordinary texel, so a sentinel value would alias onto valid data. - pub has_uv: bool, /// Width of the ray's texture footprint at this hit, in the *chart's* UV /// units — the filter width a UV texture should read `uv` with. /// @@ -87,9 +83,8 @@ impl Default for HitRecord { t: 0.0, front_face: false, face: None, - uv: (0.0, 0.0), + uv: None, tangent: Vec3A::ZERO, - has_uv: false, uv_width: 0.0, face_width: 0.0, } diff --git a/crates/crust-core/src/lib.rs b/crates/crust-core/src/lib.rs index dd8103c8..00cf589a 100644 --- a/crates/crust-core/src/lib.rs +++ b/crates/crust-core/src/lib.rs @@ -41,7 +41,7 @@ mod scene; /// crate, re-exported below as [`mtlx`]. pub use material::materialx; mod stats; -pub mod subsurface; +mod subsurface; mod texture; mod tracer; mod volume; @@ -66,13 +66,11 @@ pub use buffer::Buffer; pub use camera::Camera; pub use config::{Config, PtexMipSpace, TriPackets, config}; -/// What every kernel scene commits with — the `CRUST_TRI_PACKETS` and -/// `CRUST_BVH_PACKET_SAH` switches, read once. -pub fn commit_options() -> crust_rt::CommitOptions { - let c = config(); +/// What every kernel scene commits with — the `CRUST_TRI_PACKETS` switch, +/// read once. +pub(crate) fn commit_options() -> crust_rt::CommitOptions { crust_rt::CommitOptions { - layout: c.tri_packets.into(), - packet_sah: c.bvh_packet_sah, + layout: config().tri_packets.into(), } } pub use environment::EnvironmentMap; diff --git a/crates/crust-core/src/light/list.rs b/crates/crust-core/src/light/list.rs index 9ae88258..e8cb77f8 100644 --- a/crates/crust-core/src/light/list.rs +++ b/crates/crust-core/src/light/list.rs @@ -90,7 +90,7 @@ pub struct LightLinks { /// /// The pick's probability is half of the light strategy's MIS density (the /// other half is the light's own `sample_li` pdf), so whatever -/// [`LightList::pick`] reports, [`LightList::find_by_geom`] and +/// [`LightList::pick`] reports, [`LightList::find_index_by_geom_at`] and /// [`LightList::iter`] report the same number for the same light: the bounce /// side weights emission it found by chance with it, and the two sides must /// describe one strategy or emission is double-counted. @@ -375,7 +375,7 @@ impl LightList { } /// The light strategy's solid-angle density for a light chosen with - /// probability `pmf` (from [`LightList::pick`], [`LightList::find_by_geom`] + /// probability `pmf` (from [`LightList::pick`], [`LightList::find_index_by_geom_at`] /// or [`LightList::iter`]) whose own `sample_li` density is `light_pdf`: /// their product. Both MIS halves route through here, so they cannot /// disagree on it. Under uniform selection it is the division @@ -414,24 +414,9 @@ impl LightList { Some((index, self.pmf(index))) } - /// Finds the light whose scene geometry has world id `geom_id`, with its - /// selection probability. Used by the integrator to attribute a - /// bounce-hit emissive surface to its light for MIS; emissive geometry - /// with no light-list entry returns `None`. - pub fn find_by_geom(&self, geom_id: u32) -> Option<(&LightKind, f32)> { - let &index = self.by_geom.get(&geom_id)?; - Some((&self.lights[index], self.pmf(index))) - } - - /// [`LightList::pick`] for a vertex at `p`: under a learned selection, from - /// the distribution of the cell holding `p`; otherwise exactly `pick`. - #[inline] - pub fn pick_at(&self, p: Vec3A, u: f32) -> Option<(&LightKind, f32)> { - self.pick_index_at(p, u) - .map(|(index, pmf)| (&self.lights[index], pmf)) - } - - /// [`LightList::pick_at`] as an index into [`LightList::lights`]. + /// [`LightList::pick`] for a vertex at `p`, as an index into + /// [`LightList::lights`]: under a learned selection, from the distribution + /// of the cell holding `p`; otherwise exactly `pick_index`. /// /// Forced inline: it sits on every NEE pick, and LLVM's own threshold /// outlined it once `trace_path` grew by a few instructions elsewhere, @@ -447,7 +432,7 @@ impl LightList { } } - /// The probability [`LightList::pick_at`] at `p` picks light `index` — + /// The probability [`LightList::pick_index_at`] at `p` picks light `index` — /// what the bounce side weights emission found from a vertex at `p` with. #[inline] pub fn pmf_at(&self, p: Vec3A, index: usize) -> f32 { @@ -457,35 +442,17 @@ impl LightList { } } - /// [`LightList::find_by_geom`] with the pick probability of a vertex at - /// `p`: the bounce-side half of [`LightList::pick_at`]. - pub fn find_by_geom_at(&self, geom_id: u32, p: Vec3A) -> Option<(&LightKind, f32)> { - self.find_index_by_geom_at(geom_id, p) - .map(|(index, pmf)| (&self.lights[index], pmf)) - } - - /// [`LightList::find_by_geom_at`] as an index into [`LightList::lights`]. + /// The light whose scene geometry has world id `geom_id`, as an index into + /// [`LightList::lights`], with the pick probability of a vertex at `p`: the + /// bounce-side half of [`LightList::pick_index_at`]. Emissive geometry with + /// no light-list entry returns `None`. pub fn find_index_by_geom_at(&self, geom_id: u32, p: Vec3A) -> Option<(usize, f32)> { let &index = self.by_geom.get(&geom_id)?; Some((index, self.pmf_at(p, index))) } - /// [`LightList::iter`] with the pick probabilities of a vertex at `p`. - pub fn iter_at(&self, p: Vec3A) -> impl Iterator { - let table = self.cache.as_ref().and_then(|c| c.lookup(p)).map(|t| t.0); - self.lights.iter().enumerate().map(move |(index, light)| { - ( - light, - match table { - Some(pmf) => pmf[index], - None => self.pmf(index), - }, - ) - }) - } - - /// [`LightList::iter_at`] restricted to the lights at infinity, in the - /// same order: what an escaping ray needs. A finite light's + /// The lights at infinity with the pick probabilities of a vertex at `p`, + /// in list order: what an escaping ray needs. A finite light's /// [`Light::escaped`] is `None` by contract, so skipping it changes /// nothing but the number of virtual calls. pub fn infinite_at(&self, p: Vec3A) -> impl Iterator { @@ -506,19 +473,8 @@ impl LightList { }) } - /// [`LightList::infinite_at`] restricted to the lights an escaping ray - /// of category `mask` sees. - pub fn infinite_seen_by( - &self, - p: Vec3A, - mask: RayMask, - ) -> impl Iterator { - self.infinite_indexed_seen_by(p, mask) - .map(|(_, light, pmf)| (light, pmf)) - } - - /// [`LightList::infinite_seen_by`] with each light's index into - /// [`LightList::lights`]. + /// [`LightList::infinite_at`] restricted to the lights an escaping ray of + /// category `mask` sees, with each light's index into [`LightList::lights`]. pub fn infinite_indexed_seen_by( &self, p: Vec3A, diff --git a/crates/crust-core/src/light/tests.rs b/crates/crust-core/src/light/tests.rs index 9ca57119..bd54a822 100644 --- a/crates/crust-core/src/light/tests.rs +++ b/crates/crust-core/src/light/tests.rs @@ -330,15 +330,15 @@ fn find_by_geom_matches_by_id() { 7, )); - assert!(lights.find_by_geom(7).is_some()); - assert!(lights.find_by_geom(8).is_none()); + assert!(lights.index_of_geom(7).is_some()); + assert!(lights.index_of_geom(8).is_none()); } /// An escaping ray asks only the lights at infinity, and must get exactly /// what asking every light would: the same lights with the same pick /// probabilities, in list order, under both uniform and power selection. #[test] -fn infinite_at_is_iter_at_filtered_to_escaped() { +fn infinite_at_is_every_escaping_light() { let mat = Arc::new(Emissive::new(Vec3A::splat(1.0))); let mut lights = LightList::new(); let area = |c: f32, id: u32| { @@ -359,11 +359,9 @@ fn infinite_at_is_iter_at_filtered_to_escaped() { lights.select_by(selection); let from = Vec3A::new(0.0, 5.0, 0.0); let dir = Vec3A::Y; - let every: Vec<(usize, f32)> = lights - .iter_at(from) - .enumerate() - .filter(|(_, (l, _))| l.escaped(from, dir).is_some()) - .map(|(i, (_, pmf))| (i, pmf)) + let every: Vec<(usize, f32)> = (0..lights.count()) + .filter(|&i| lights.light(i).escaped(from, dir).is_some()) + .map(|i| (i, lights.pmf_at(from, i))) .collect(); let infinite: Vec = lights.infinite_at(from).map(|(_, pmf)| pmf).collect(); assert_eq!(every.iter().map(|&(i, _)| i).collect::>(), [1, 3]); @@ -444,14 +442,14 @@ fn remove_is_as_if_never_added() { l.select_by(LightSelection::Power); } assert_eq!(removed.count(), never.count()); - assert!(removed.find_by_geom(0).is_none()); + assert!(removed.index_of_geom(0).is_none()); assert_eq!( - removed.find_by_geom(1).map(|(_, p)| p.to_bits()), - never.find_by_geom(1).map(|(_, p)| p.to_bits()) + removed.index_of_geom(1).map(|i| removed.pmf(i).to_bits()), + never.index_of_geom(1).map(|i| never.pmf(i).to_bits()) ); let seen = |l: &LightList| { - l.infinite_seen_by(Vec3A::ZERO, MASK_CAMERA) - .map(|(_, p)| p.to_bits()) + l.infinite_indexed_seen_by(Vec3A::ZERO, MASK_CAMERA) + .map(|(_, _, p)| p.to_bits()) .collect::>() }; assert_eq!(seen(&removed), seen(&never)); diff --git a/crates/crust-core/src/material/closure/mod.rs b/crates/crust-core/src/material/closure/mod.rs index af07c07f..f30e3a4a 100644 --- a/crates/crust-core/src/material/closure/mod.rs +++ b/crates/crust-core/src/material/closure/mod.rs @@ -136,7 +136,7 @@ pub enum Lobe { }, /// `subsurface_bsdf`: no value toward any direction (as in Typhoon, the /// leaf does no NEE); selecting it enters a random walk - /// ([`crate::subsurface`]) through the interface above it. + /// (`crust_core`'s `subsurface` module) through the interface above it. Subsurface { color: Vec3A, radius: Vec3A, diff --git a/crates/crust-core/src/material/material.rs b/crates/crust-core/src/material/material.rs index a725b0d5..3d9d39c2 100644 --- a/crates/crust-core/src/material/material.rs +++ b/crates/crust-core/src/material/material.rs @@ -35,7 +35,7 @@ pub struct ScatterSample { pub spread: f32, /// Set when the material selected a subsurface leaf: the direction is /// not a bounce but the entry into a random walk - /// ([`crate::subsurface`]), which the tracer runs before the path + /// (`crust_core`'s `subsurface` module), which the tracer runs before the path /// resumes at the walk's exit. Such a sample is `delta` — no continuous /// density can produce it — and its `value` is the leaf's weight over its /// selection probability. The value is the leaf's index, which diff --git a/crates/crust-core/src/material/materialx.rs b/crates/crust-core/src/material/materialx.rs index 681b60a7..0aead262 100644 --- a/crates/crust-core/src/material/materialx.rs +++ b/crates/crust-core/src/material/materialx.rs @@ -272,7 +272,7 @@ fn run( f: impl FnOnce(&[Val]) -> R, ) -> R { let ctx = ShadeCtx { - uv: if rec.has_uv { rec.uv } else { (0.0, 0.0) }, + uv: rec.uv.unwrap_or((0.0, 0.0)), normal: rec.normal, tangent: rec.tangent, view: r_in.direction(), diff --git a/crates/crust-core/src/material/preview_surface.rs b/crates/crust-core/src/material/preview_surface.rs index 838662ed..bf33988c 100644 --- a/crates/crust-core/src/material/preview_surface.rs +++ b/crates/crust-core/src/material/preview_surface.rs @@ -148,7 +148,7 @@ impl UvInput { /// The four output channels at a hit, `scale`/`bias` applied. #[inline] fn sample(&self, rec: &HitRecord) -> [f32; 4] { - self.sample_at(rec.has_uv.then_some(rec.uv), rec.uv_width) + self.sample_at(rec.uv, rec.uv_width) } /// The four output channels at chart point `uv` over a footprint `width` @@ -534,8 +534,7 @@ mod tests { HitRecord { normal: Vec3A::Z, tangent: Vec3A::X, - uv: (u, 0.5), - has_uv: true, + uv: Some((u, 0.5)), ..HitRecord::default() } } diff --git a/crates/crust-core/src/rt_world.rs b/crates/crust-core/src/rt_world.rs index b508dbcc..4e39de28 100644 --- a/crates/crust-core/src/rt_world.rs +++ b/crates/crust-core/src/rt_world.rs @@ -766,20 +766,20 @@ impl World { m.resolve(h.prim_id, h.u, h.v, tables.swapped) .map(|(id, u, v)| crate::hittable::FaceHit { id, uv: (u, v) }) }); - let (uv, tangent, has_uv) = match &tables.uv { + let (uv, tangent) = match &tables.uv { Some(m) => { let verts = self.world_vertices(h.geom_id, h.prim_id, tables); match m.resolve(h.prim_id, h.u, h.v, tables.swapped, verts) { - Some((uv, tangent)) => (uv, tangent, true), - None => ((0.0, 0.0), Vec3A::ZERO, false), + Some((uv, tangent)) => (Some(uv), tangent), + None => (None, Vec3A::ZERO), } } // No UV map: a curve's direction, which the kernel carries to // world space through every placement (zero for anything else) — // the strand a fibre BSDF shades along. Tested before normalising: // nearly every hit has none, and the square root is not free. - None if h.dpdu == Vec3A::ZERO => ((0.0, 0.0), Vec3A::ZERO, false), - None => ((0.0, 0.0), h.dpdu.normalize_or_zero(), false), + None if h.dpdu == Vec3A::ZERO => (None, Vec3A::ZERO), + None => (None, h.dpdu.normalize_or_zero()), }; // The ray's texture footprint, converted into each parameterisation // the shader might index. Zero unless the ray carries a cone *and* @@ -814,7 +814,6 @@ impl World { face, uv, tangent, - has_uv, uv_width, face_width, }, diff --git a/crates/crust-core/src/scene/usd_import/mesh.rs b/crates/crust-core/src/scene/usd_import/mesh.rs index cba3a5f3..90bab1b2 100644 --- a/crates/crust-core/src/scene/usd_import/mesh.rs +++ b/crates/crust-core/src/scene/usd_import/mesh.rs @@ -1496,6 +1496,10 @@ fn check_face_count(prim: &Prim, n_base_faces: usize, material: &dyn Material) { } } +/// The triangle list, plus the per-face and per-triangle-UV side tables +/// when the material asked for them. +type Triangulated = (Vec<[u32; 3]>, Option, Option); + /// Fan-triangulates the faces into an index-triple list; `None` if /// nothing survives. /// @@ -1505,11 +1509,6 @@ fn check_face_count(prim: &Prim, n_base_faces: usize, material: &dyn Material) { /// outputs are index-parallel by construction: every `push` to one pushes to /// the other in the same statement, so the skip paths cannot desynchronise /// them. -/// -/// The triangle list, plus the per-face and per-triangle-UV side tables -/// when the material asked for them. -type Triangulated = (Vec<[u32; 3]>, Option, Option); - fn triangulate( counts: &[i32], indices: &[i32], diff --git a/crates/crust-core/src/scene/usd_import/settings.rs b/crates/crust-core/src/scene/usd_import/settings.rs index b76c8057..0c1e954f 100644 --- a/crates/crust-core/src/scene/usd_import/settings.rs +++ b/crates/crust-core/src/scene/usd_import/settings.rs @@ -14,13 +14,6 @@ use crate::tracer::{RenderSettings, SamplingStrategy}; use super::attrs::{custom_bool, custom_f32, custom_i32, custom_token, value_at}; use super::prim_at; -const DEFAULT_SPP: u32 = 128; -const DEFAULT_MAX_DEPTH: u32 = 32; -const DEFAULT_WIDTH: usize = 640; -const DEFAULT_HEIGHT: usize = 360; -const DEFAULT_MIN_SPP: u32 = 32; -const DEFAULT_VARIANCE: f32 = 0.05; -const DEFAULT_FRAME: isize = 0; const DEFAULT_GUIDING_TRAIN_ITERATIONS: u32 = 4; const DEFAULT_GUIDING_PROB: f32 = 0.5; /// Hydra's default for `domeLightCameraVisibility`: the camera sees domes. @@ -146,7 +139,8 @@ pub(super) fn import_render_settings(stage: &Stage) -> RenderSettings { } }; - let (mut w, mut h) = (DEFAULT_WIDTH, DEFAULT_HEIGHT); + let d = RenderSettings::default(); + let (mut w, mut h) = d.get_dimensions(); if let Some(v2) = value_at(&s.resolution_attr()).and_then(|v| v.try_as_vec_2i()) { w = v2.x as usize; h = v2.y as usize; @@ -154,20 +148,22 @@ pub(super) fn import_render_settings(stage: &Stage) -> RenderSettings { // Custom `crust:*` attrs. We look them up on the RenderSettings prim. let prim = prim_at(stage, path); - let spp = custom_i32(&prim, "crust:samplesPerPixel").unwrap_or(DEFAULT_SPP as i32) as u32; - let max_depth = custom_i32(&prim, "crust:maxDepth").unwrap_or(DEFAULT_MAX_DEPTH as i32) as u32; + let spp = + custom_i32(&prim, "crust:samplesPerPixel").map_or(d.samples_per_pixel(), |n| n as u32); + let max_depth = custom_i32(&prim, "crust:maxDepth").map_or(d.max_depth(), |n| n as u32); // A negative minimum is refused rather than cast: `-1 as u32` is // `u32::MAX`, which would overflow the first check point. let min_spp = match custom_i32(&prim, "crust:minSamplesPerPixel") { Some(n) if n < 0 => { - warn!("crust:minSamplesPerPixel = {n} is negative — using {DEFAULT_MIN_SPP}"); - DEFAULT_MIN_SPP + let fallback = d.min_samples_per_pixel(); + warn!("crust:minSamplesPerPixel = {n} is negative — using {fallback}"); + fallback } Some(n) => n as u32, - None => DEFAULT_MIN_SPP, + None => d.min_samples_per_pixel(), }; - let variance = custom_f32(&prim, "crust:varianceThreshold").unwrap_or(DEFAULT_VARIANCE); - let frame = custom_i32(&prim, "crust:frame").unwrap_or(DEFAULT_FRAME as i32) as isize; + let variance = custom_f32(&prim, "crust:varianceThreshold").unwrap_or(d.variance_threshold()); + let frame = custom_i32(&prim, "crust:frame").map_or(d.frame(), |n| n as isize); // Path guiding (opt-in). let guiding = custom_bool(&prim, "crust:pathGuiding").unwrap_or(false); @@ -250,7 +246,11 @@ pub(super) fn import_render_settings(stage: &Stage) -> RenderSettings { "off".to_string() } ); - RenderSettings::new(spp, max_depth, w, h, min_spp, variance, frame) + d.with_resolution(w, h) + .with_max_depth(max_depth) + .with_adaptive_sampling(min_spp, variance) + .with_frame(frame) + .with_samples_per_pixel(spp) .with_guiding(guiding, guiding_iters, guiding_prob) .with_sampling_strategy(strategy) .with_light_selection(light_selection) @@ -281,13 +281,5 @@ pub(super) fn check_time_range(stage: &Stage, time: f64) { } fn default_settings() -> RenderSettings { - RenderSettings::new( - DEFAULT_SPP, - DEFAULT_MAX_DEPTH, - DEFAULT_WIDTH, - DEFAULT_HEIGHT, - DEFAULT_MIN_SPP, - DEFAULT_VARIANCE, - DEFAULT_FRAME, - ) + RenderSettings::default() } diff --git a/crates/crust-core/src/stats.rs b/crates/crust-core/src/stats.rs index f4a5627b..40086ecf 100644 --- a/crates/crust-core/src/stats.rs +++ b/crates/crust-core/src/stats.rs @@ -262,39 +262,69 @@ impl RayStats { /// Sums another unit's counters into this one. pub fn merge(&mut self, o: &RayStats) { + // Destructured, not read field by field: a counter added to the + // struct and forgotten here is a compile error, not a silent zero in + // every report. + let RayStats { + camera_rays, + closest_hit, + shadow_rays, + vertices, + rr_tested, + rr_killed, + ended_escaped, + ended_depth, + ended_absorbed, + volume_scatters, + medium_scatters, + sss_walks, + sss_exits, + sss_steps, + sss_rays, + cutout_passes, + cutout_rays, + light_samples, + shadow_occluded, + adaptive_pixels, + adaptive_samples, + early_stopped, + spp_min, + spp_max, + neighbour_held, + } = *o; // Min over pixels actually counted: a unit with no adaptive pixel has // `spp_min == 0`, which is not a pixel that took zero samples. - if o.adaptive_pixels > 0 { + if adaptive_pixels > 0 { self.spp_min = if self.adaptive_pixels == 0 { - o.spp_min + spp_min } else { - self.spp_min.min(o.spp_min) + self.spp_min.min(spp_min) }; } - self.spp_max = self.spp_max.max(o.spp_max); - self.adaptive_pixels += o.adaptive_pixels; - self.adaptive_samples += o.adaptive_samples; - self.early_stopped += o.early_stopped; - self.neighbour_held += o.neighbour_held; - self.ended_absorbed += o.ended_absorbed; - self.volume_scatters += o.volume_scatters; - self.medium_scatters += o.medium_scatters; - self.sss_walks += o.sss_walks; - self.sss_exits += o.sss_exits; - self.sss_steps += o.sss_steps; - self.sss_rays += o.sss_rays; - self.cutout_passes += o.cutout_passes; - self.cutout_rays += o.cutout_rays; - self.light_samples += o.light_samples; - self.shadow_occluded += o.shadow_occluded; - self.camera_rays += o.camera_rays; - self.closest_hit += o.closest_hit; - self.shadow_rays += o.shadow_rays; - self.vertices += o.vertices; - self.rr_tested += o.rr_tested; - self.rr_killed += o.rr_killed; - self.ended_escaped += o.ended_escaped; - self.ended_depth += o.ended_depth; + self.spp_max = self.spp_max.max(spp_max); + self.adaptive_pixels += adaptive_pixels; + self.adaptive_samples += adaptive_samples; + self.early_stopped += early_stopped; + self.neighbour_held += neighbour_held; + self.ended_absorbed += ended_absorbed; + self.volume_scatters += volume_scatters; + self.medium_scatters += medium_scatters; + self.sss_walks += sss_walks; + self.sss_exits += sss_exits; + self.sss_steps += sss_steps; + self.sss_rays += sss_rays; + self.cutout_passes += cutout_passes; + self.cutout_rays += cutout_rays; + self.light_samples += light_samples; + self.shadow_occluded += shadow_occluded; + self.camera_rays += camera_rays; + self.closest_hit += closest_hit; + self.shadow_rays += shadow_rays; + self.vertices += vertices; + self.rr_tested += rr_tested; + self.rr_killed += rr_killed; + self.ended_escaped += ended_escaped; + self.ended_depth += ended_depth; } fn is_empty(&self) -> bool { diff --git a/crates/crust-core/src/tracer/path.rs b/crates/crust-core/src/tracer/path.rs index 652e1cb8..67cb1326 100644 --- a/crates/crust-core/src/tracer/path.rs +++ b/crates/crust-core/src/tracer/path.rs @@ -35,7 +35,7 @@ pub(super) const K_TIME: i32 = 2; // off root: shutter time for motion blur const K_NEE: i32 = 0; // off vertex: light pick (0) + area uv (1,2) const K_NEE_SHADOW: i32 = 1; // off vertex: shadow-ray volume transmittance const K_BSDF: i32 = 2; // off vertex: material scatter block -const K_GUIDE: i32 = 3; // off vertex: guide coin (0) + guide seed (1,2) +const K_GUIDE: i32 = 3; // off vertex: guide coin (0) + its rng() for the descent const K_PHASE: i32 = 4; // off vertex: phase lobe (0) + HG uv (1,2) const K_RR: i32 = 5; // off vertex: Russian-roulette survival const K_MEDIUM: i32 = 6; // off vertex: carried-medium free flight @@ -118,6 +118,15 @@ pub fn ray_color( ) } +/// Are texture-filtering ray cones on? `CRUST_RAY_CONES=0` forces every +/// footprint to zero, which makes every texture point-sample its finest level +/// — the A/B that separates "the mip pyramids changed the image" from "the +/// footprints did". Consulted per camera ray, so it reads the parsed +/// [`crate::config()`], never the environment. +pub(super) fn ray_cones_enabled() -> bool { + crate::config().ray_cones +} + /// Choose the bounce direction and the pdf its contribution is divided by. /// /// With guiding this is one-sample MIS between the guiding distribution and @@ -129,15 +138,7 @@ pub fn ray_color( /// guide can never produce: they keep their placeholder pdf, are never mixed /// with a continuous density, and their value is divided by `1-α` to /// compensate for the coin reducing the delta lobe's selection probability. -/// Are texture-filtering ray cones on? `CRUST_RAY_CONES=0` forces every -/// footprint to zero, which makes every texture point-sample its finest level -/// — the A/B that separates "the mip pyramids changed the image" from "the -/// footprints did". Consulted per camera ray, so it reads the parsed -/// [`crate::config()`], never the environment. -pub(super) fn ray_cones_enabled() -> bool { - crate::config().ray_cones -} - +// // `inline(always)`, as is `escaped_emission`: each is called once per // `trace_path` instance, and once the integrator was monomorphised on the // profiler switch LLVM stopped inlining them into either copy — +1.1% @@ -160,19 +161,21 @@ fn sample_bounce_direction( None => sp.scatter_importance(r, dom), }; // Distinct sub-domains: the BSDF scatter block, and the guide block whose - // first dimension is the α-coin and next two are the guide-sampling seed. + // first dimension is the α-coin and whose incidental stream drives the + // quadtree descent. let bsdf_dom = sampler.new_domain(K_BSDF); let g = match guiding { Some(g) if g.field.trained_at(rec.p) => g, _ => return scatter(bsdf_dom), }; let alpha = g.field.config().guide_prob; - let gs = sampler.new_domain(K_GUIDE).draw_sample_f32::<4>(); + let guide_dom = sampler.new_domain(K_GUIDE); + let gs = guide_dom.draw_sample_f32::<1>(); if gs[0] < alpha { // Guide branch: draw from the field; the material's continuous // component supplies the value and the BSDF side of the mixture pdf. - if let Some((wi, p_guide)) = g.field.sample(rec.p, [gs[1], gs[2]]) + if let Some((wi, p_guide)) = g.field.sample(rec.p, &mut guide_dom.rng()) && let Some((value, p_bsdf)) = sp.eval(r, wi) { if AOV && let Some(out) = split { @@ -1362,7 +1365,7 @@ pub(super) fn trace_path( *first = FirstHit::Surface { p: rec.p, n: sp.normal(), - uv: rec.has_uv.then_some(rec.uv), + uv: rec.uv, }; } let emitted = sp.emitted(); diff --git a/crates/crust-core/src/tracer/settings.rs b/crates/crust-core/src/tracer/settings.rs index a71851c2..ac188e57 100644 --- a/crates/crust-core/src/tracer/settings.rs +++ b/crates/crust-core/src/tracer/settings.rs @@ -159,25 +159,22 @@ pub struct RenderSettings { // 0). Validated at construction: `Some` is always finite and positive. pub(super) indirect_clamp: Option, } -impl RenderSettings { - pub fn new( - samples_per_pixel: u32, - max_depth: u32, - width: usize, - height: usize, - min_samples_per_pixel: u32, - variance_threshold: f32, - frame: isize, - ) -> Self { +/// The settings a stage that authors none renders with: 640×360 at 128 spp, +/// paths up to 32 vertices, adaptive sampling stopping no earlier than 32 +/// samples at a 5% relative standard error, frame 0. Every other field is +/// changed by name through a `with_*` builder, so no call site passes a row +/// of bare numbers whose order only the signature knows. +impl Default for RenderSettings { + fn default() -> Self { RenderSettings { - samples_per_pixel, - max_depth, - width, - height, - min_samples_per_pixel, - variance_threshold, + samples_per_pixel: 128, + max_depth: 32, + width: 640, + height: 360, + min_samples_per_pixel: 32, + variance_threshold: 0.05, adaptive_neighbour_tolerance: DEFAULT_ADAPTIVE_NEIGHBOUR_TOLERANCE, - frame, + frame: 0, guiding: false, guiding_train_iterations: 4, guiding_prob: 0.5, @@ -187,15 +184,35 @@ impl RenderSettings { indirect_clamp: Some(DEFAULT_INDIRECT_CLAMP), } } +} - /// Override the image resolution — the first `RenderProduct`'s, when it - /// authors its own. - pub(crate) fn with_resolution(mut self, width: usize, height: usize) -> Self { +impl RenderSettings { + /// Set the image resolution, in pixels. + pub fn with_resolution(mut self, width: usize, height: usize) -> Self { self.width = width; self.height = height; self } + /// Set the longest path, in vertices. + pub fn with_max_depth(mut self, max_depth: u32) -> Self { + self.max_depth = max_depth; + self + } + + /// Set adaptive sampling: a pixel may stop once it has taken at least + /// `min_samples_per_pixel` samples and the relative standard error of its + /// mean is below `variance_threshold` (`0` never stops early). + pub fn with_adaptive_sampling( + mut self, + min_samples_per_pixel: u32, + variance_threshold: f32, + ) -> Self { + self.min_samples_per_pixel = min_samples_per_pixel; + self.variance_threshold = variance_threshold; + self + } + /// Override the samples-per-pixel count (e.g. from a CLI flag). Clamped to >= 1. pub fn with_samples_per_pixel(mut self, spp: u32) -> Self { self.samples_per_pixel = spp.max(1); diff --git a/crates/crust-core/src/world.rs b/crates/crust-core/src/world.rs index 838b7a40..dc6c5679 100644 --- a/crates/crust-core/src/world.rs +++ b/crates/crust-core/src/world.rs @@ -124,7 +124,11 @@ pub fn get_settings() -> (Camera, RenderSettings) { aperture, dist_to_focus, ); - let render_settings = RenderSettings::new(64, 32, IMAGE_WIDTH, IMAGE_HEIGHT, 32, 0.05, 0); + let render_settings = RenderSettings::default() + .with_resolution(IMAGE_WIDTH, IMAGE_HEIGHT) + .with_samples_per_pixel(64) + .with_max_depth(32) + .with_adaptive_sampling(32, 0.05); (cam, render_settings) } diff --git a/crates/crust-core/tests/aovs.rs b/crates/crust-core/tests/aovs.rs index abc3ed1a..0a7e297b 100644 --- a/crates/crust-core/tests/aovs.rs +++ b/crates/crust-core/tests/aovs.rs @@ -95,8 +95,12 @@ fn scene(spp: u32, variance: f32, guiding: bool, wall: bool) -> Renderer { // One training iteration: with two or more, whether the final pass is // guided depends on their wall-clock efficiency (`ΔEff`), and a render // is not repeatable at all — let alone comparable with another one. - let settings = - RenderSettings::new(spp, 4, W, H, spp.min(8), variance, 0).with_guiding(guiding, 1, 0.5); + let settings = RenderSettings::default() + .with_resolution(W, H) + .with_samples_per_pixel(spp) + .with_max_depth(4) + .with_adaptive_sampling(spp.min(8), variance) + .with_guiding(guiding, 1, 0.5); Renderer::new(camera, world.commit(), lights, settings) } diff --git a/crates/crust-core/tests/camera_buffer_ray.rs b/crates/crust-core/tests/camera_buffer_ray.rs index e7648308..76699141 100644 --- a/crates/crust-core/tests/camera_buffer_ray.rs +++ b/crates/crust-core/tests/camera_buffer_ray.rs @@ -112,8 +112,7 @@ fn ray_builders_and_kernel_view_agree() { fn hit_record_defaults_to_no_face_and_no_uv() { let h = HitRecord::new(); assert_eq!(h.face, None); - assert!(!h.has_uv); - assert_eq!(h.uv, (0.0, 0.0)); + assert!(h.uv.is_none()); assert_eq!(h.tangent, Vec3A::ZERO); assert!(!h.front_face); assert_eq!(h.t, 0.0); diff --git a/crates/crust-core/tests/guiding.rs b/crates/crust-core/tests/guiding.rs index 3718456a..86e68186 100644 --- a/crates/crust-core/tests/guiding.rs +++ b/crates/crust-core/tests/guiding.rs @@ -22,11 +22,12 @@ fn sample_scene() -> PathBuf { fn render_lum(guided: bool, spp: u32, train_iterations: u32) -> Vec { const RES: usize = 96; let scene = Scene::from_usd(&sample_scene()).expect("load cornellbox_guided.usda"); - let settings = RenderSettings::new(spp, 8, RES, RES, 16, 0.05, 0).with_guiding( - guided, - train_iterations, - 0.5, - ); + let settings = RenderSettings::default() + .with_resolution(RES, RES) + .with_samples_per_pixel(spp) + .with_max_depth(8) + .with_adaptive_sampling(16, 0.05) + .with_guiding(guided, train_iterations, 0.5); let renderer = Renderer::new(scene.camera, scene.world, scene.lights, settings); let buf = renderer.render(); let mut out = Vec::with_capacity(RES * RES); @@ -64,7 +65,12 @@ fn guided_render_is_unbiased() { #[ignore = "renders a frame; run explicitly with --ignored"] fn guided_render_with_tiles_smoke() { let scene = Scene::from_usd(&sample_scene()).expect("load cornellbox_guided.usda"); - let settings = RenderSettings::new(8, 6, 64, 64, 4, 0.05, 0).with_guiding(true, 3, 0.5); + let settings = RenderSettings::default() + .with_resolution(64, 64) + .with_samples_per_pixel(8) + .with_max_depth(6) + .with_adaptive_sampling(4, 0.05) + .with_guiding(true, 3, 0.5); let renderer = Renderer::new(scene.camera, scene.world, scene.lights, settings); let buf = renderer.render_with_tiles(); // The closed box is lit: the image cannot be black. @@ -88,7 +94,12 @@ fn guided_tiles_and_rows_are_bit_identical() { let (w, h) = (37, 21); let render = |tiled: bool| { let scene = Scene::from_usd(&sample_scene()).expect("load cornellbox_guided.usda"); - let settings = RenderSettings::new(8, 6, w, h, 4, 0.05, 0).with_guiding(true, 3, 0.5); + let settings = RenderSettings::default() + .with_resolution(w, h) + .with_samples_per_pixel(8) + .with_max_depth(6) + .with_adaptive_sampling(4, 0.05) + .with_guiding(true, 3, 0.5); let renderer = Renderer::new(scene.camera, scene.world, scene.lights, settings); if tiled { renderer.render_with_tiles() diff --git a/crates/crust-core/tests/guiding_field.rs b/crates/crust-core/tests/guiding_field.rs index 6ba3ee11..52b70219 100644 --- a/crates/crust-core/tests/guiding_field.rs +++ b/crates/crust-core/tests/guiding_field.rs @@ -49,12 +49,15 @@ fn an_untrained_field_has_nothing_to_sample() { for _ in 0..20 { let p = Vec3A::new(rng.next_f32(), rng.next_f32(), rng.next_f32()); assert!(!f.trained_at(p)); - assert!(f.sample(p, rng.next_2d()).is_none()); + assert!(f.sample(p, &mut rng).is_none()); assert_eq!(f.pdf(p, Vec3A::Z), 0.0); } // Points outside the bounds are handled, not panicked on. assert!(!f.trained_at(Vec3A::splat(5.0))); - assert!(f.sample(Vec3A::splat(-5.0), [0.5, 0.5]).is_none()); + assert!( + f.sample(Vec3A::splat(-5.0), &mut openqmc::pcg::Rng::new(1)) + .is_none() + ); } #[test] @@ -65,7 +68,7 @@ fn training_makes_the_field_sampleable() { let p = Vec3A::splat(0.5); assert!(f.trained_at(p)); for _ in 0..50 { - let (d, pdf) = f.sample(p, rng.next_2d()).expect("trained"); + let (d, pdf) = f.sample(p, &mut rng).expect("trained"); assert!((d.length() - 1.0).abs() < 1e-3, "{d}"); assert!(pdf > 0.0 && pdf.is_finite()); } @@ -83,7 +86,7 @@ fn a_trained_field_concentrates_on_the_taught_direction() { let p = Vec3A::splat(0.5); let aligned_after = |f: &GuidingField, rng: &mut Rng| { (0..500) - .filter(|_| f.sample(p, rng.next_2d()).unwrap().0.dot(target) > 0.5) + .filter(|_| f.sample(p, rng).unwrap().0.dot(target) > 0.5) .count() }; f.update(&samples_toward(target, 4000, 1.0), 1); @@ -120,7 +123,7 @@ fn sample_and_pdf_agree() { f.update(&data, 1); let p = Vec3A::splat(0.5); for _ in 0..200 { - let (d, pdf) = f.sample(p, rng.next_2d()).unwrap(); + let (d, pdf) = f.sample(p, &mut rng).unwrap(); let again = f.pdf(p, d); assert!( (again - pdf).abs() < 1e-3 * pdf.max(1.0), @@ -204,7 +207,10 @@ fn zero_radiance_samples_do_not_train() { let mut f = GuidingField::new(unit_bounds(), GuidingConfig::default()); f.update(&samples_toward(Vec3A::Y, 500, 0.0), 1); assert!(!f.trained_at(Vec3A::splat(0.5))); - assert!(f.sample(Vec3A::splat(0.5), [0.3, 0.3]).is_none()); + assert!( + f.sample(Vec3A::splat(0.5), &mut openqmc::pcg::Rng::new(1)) + .is_none() + ); } #[test] @@ -246,5 +252,8 @@ fn degenerate_bounds_are_padded() { .collect(); f.update(&data, 1); assert!(f.trained_at(Vec3A::new(0.5, 0.0, 0.5))); - assert!(f.sample(Vec3A::new(0.5, 0.0, 0.5), [0.2, 0.7]).is_some()); + assert!( + f.sample(Vec3A::new(0.5, 0.0, 0.5), &mut openqmc::pcg::Rng::new(1)) + .is_some() + ); } diff --git a/crates/crust-core/tests/hair.rs b/crates/crust-core/tests/hair.rs index 37ebe408..428a8819 100644 --- a/crates/crust-core/tests/hair.rs +++ b/crates/crust-core/tests/hair.rs @@ -273,9 +273,8 @@ fn a_mixed_fibre_keeps_both_leaves() { t: 1.0, front_face: true, face: None, - uv: (0.0, 0.0), + uv: None, tangent: Vec3A::X, - has_uv: false, uv_width: 0.0, face_width: 0.0, }; diff --git a/crates/crust-core/tests/learned_selection.rs b/crates/crust-core/tests/learned_selection.rs index c9b80c4d..46f3cd17 100644 --- a/crates/crust-core/tests/learned_selection.rs +++ b/crates/crust-core/tests/learned_selection.rs @@ -77,7 +77,12 @@ fn scene(selection: LightSelection, spp: u32) -> (Renderer, LightList) { 0.0, 8.0, ); - let settings = RenderSettings::new(spp, 4, W, H, spp, 0.0, 0).with_light_selection(selection); + let settings = RenderSettings::default() + .with_resolution(W, H) + .with_samples_per_pixel(spp) + .with_max_depth(4) + .with_adaptive_sampling(spp, 0.0) + .with_light_selection(selection); ( Renderer::new(camera, world.commit(), lights, settings), by_power, @@ -101,28 +106,20 @@ fn a_sealed_light_is_learned_down_to_the_defensive_share() { // The bounce side reads the same numbers the pick reports. for index in 0..2 { let geom = r.lights.lights()[index].geom_id().unwrap(); - let (_, pmf) = r.lights.find_by_geom_at(geom, p).unwrap(); + let (found, pmf) = r.lights.find_index_by_geom_at(geom, p).unwrap(); + assert_eq!(found, index); assert_eq!(pmf, r.lights.pmf_at(p, index)); - let (_, it) = r.lights.iter_at(p).nth(index).unwrap(); - assert_eq!(it, pmf); } // And the pick lands on each light as often as its pmf says. let n = 10_000; let picked_sealed = (0..n) .filter(|&i| { - let (light, pmf) = r.lights.pick_at(p, (i as f32 + 0.5) / n as f32).unwrap(); - assert_eq!( - pmf, - r.lights.pmf_at( - p, - if std::ptr::eq(light, &r.lights.lights()[0]) { - 0 - } else { - 1 - } - ) - ); - std::ptr::eq(light, &r.lights.lights()[1]) + let (index, pmf) = r + .lights + .pick_index_at(p, (i as f32 + 0.5) / n as f32) + .unwrap(); + assert_eq!(pmf, r.lights.pmf_at(p, index)); + index == 1 }) .count(); assert!((picked_sealed as f32 / n as f32 - sealed).abs() < 1e-3); @@ -232,8 +229,12 @@ fn one_light_learns_nothing() { 0.0, 8.0, ); - let settings = - RenderSettings::new(1, 2, 8, 8, 1, 0.0, 0).with_light_selection(LightSelection::Learned); + let settings = RenderSettings::default() + .with_resolution(8, 8) + .with_samples_per_pixel(1) + .with_max_depth(2) + .with_adaptive_sampling(1, 0.0) + .with_light_selection(LightSelection::Learned); let r = Renderer::new(camera, world.commit(), lights, settings); assert_eq!(r.lights.selection(), LightSelection::Power); } diff --git a/crates/crust-core/tests/lights.rs b/crates/crust-core/tests/lights.rs index 57868eba..f8ce2f58 100644 --- a/crates/crust-core/tests/lights.rs +++ b/crates/crust-core/tests/lights.rs @@ -1106,7 +1106,7 @@ fn empty_light_list_picks_nothing() { assert_eq!(l.count(), 0); assert!(l.pick(0.0).is_none()); assert!(l.pick(0.99).is_none()); - assert!(l.find_by_geom(0).is_none()); + assert!(l.index_of_geom(0).is_none()); let d = LightList::default(); assert_eq!(d.count(), 0); } @@ -1229,8 +1229,8 @@ fn power_selection_picks_by_power_defensively() { ); } for i in [0u32, 1, 2] { - let (_, found) = l.find_by_geom(10 + i).unwrap(); - assert_eq!(found, l.pmf(i as usize), "find_by_geom disagrees with pmf"); + let index = l.index_of_geom(10 + i).unwrap(); + assert_eq!(index, i as usize, "index_of_geom disagrees with the list"); } let pmfs: Vec = l.iter().map(|(_, p)| p).collect(); assert_eq!(pmfs, (0..4).map(|i| l.pmf(i)).collect::>()); @@ -1311,13 +1311,13 @@ fn light_list_finds_lights_by_geometry_id() { l.add(sphere_light(Vec3A::ZERO, 1.0, Vec3A::ONE, 12)); l.add(DistantLight::new(-Vec3A::Y, Vec3A::ONE, 1.0)); l.add(sphere_light(Vec3A::ZERO, 1.0, Vec3A::ONE, 40)); - assert!(l.find_by_geom(12).is_some()); - assert!(l.find_by_geom(40).is_some()); + assert!(l.index_of_geom(12).is_some()); + assert!(l.index_of_geom(40).is_some()); assert!( - l.find_by_geom(13).is_none(), + l.index_of_geom(13).is_none(), "an unrelated geometry is not a light" ); - assert_eq!(l.find_by_geom(40).unwrap().0.geom_id(), Some(40)); + assert_eq!(l.light(l.index_of_geom(40).unwrap()).geom_id(), Some(40)); } #[test] diff --git a/crates/crust-core/tests/lpe.rs b/crates/crust-core/tests/lpe.rs index 4b289db9..374fb552 100644 --- a/crates/crust-core/tests/lpe.rs +++ b/crates/crust-core/tests/lpe.rs @@ -181,7 +181,11 @@ fn scene(o: &Opts) -> Renderer { 0.0, 5.0, ); - let settings = RenderSettings::new(o.spp, o.depth, W, H, o.spp, 0.0, 0) + let settings = RenderSettings::default() + .with_resolution(W, H) + .with_samples_per_pixel(o.spp) + .with_max_depth(o.depth) + .with_adaptive_sampling(o.spp, 0.0) .with_indirect_clamp(o.clamp) .with_sampling_strategy(o.strategy) .with_guiding(o.guiding, 1, 0.5) @@ -645,7 +649,12 @@ fn checker_scene(spp: u32) -> Renderer { 0.0, 5.0, ); - let settings = RenderSettings::new(spp, 1, W, H, spp, 0.0, 0).with_indirect_clamp(0.0); + let settings = RenderSettings::default() + .with_resolution(W, H) + .with_samples_per_pixel(spp) + .with_max_depth(1) + .with_adaptive_sampling(spp, 0.0) + .with_indirect_clamp(0.0); Renderer::new(camera, world.commit(), lights, settings) } diff --git a/crates/crust-core/tests/mtlx_surfaces.rs b/crates/crust-core/tests/mtlx_surfaces.rs index fd962223..8dabe27b 100644 --- a/crates/crust-core/tests/mtlx_surfaces.rs +++ b/crates/crust-core/tests/mtlx_surfaces.rs @@ -47,9 +47,8 @@ fn hit(u: f32, v: f32) -> HitRecord { t: 1.0, front_face: true, face: None, - uv: (u, v), + uv: Some((u, v)), tangent: Vec3A::X, - has_uv: true, uv_width: 0.0, face_width: 0.0, } diff --git a/crates/crust-core/tests/profile.rs b/crates/crust-core/tests/profile.rs index 1bf941d9..96bd958e 100644 --- a/crates/crust-core/tests/profile.rs +++ b/crates/crust-core/tests/profile.rs @@ -12,7 +12,11 @@ fn renderer() -> Renderer { let (world, lights) = simple_scene(); let (camera, _) = get_settings(); // Adaptive stop off, so both renders take exactly the same samples. - let settings = RenderSettings::new(4, 4, W, H, 4, 0.0, 0); + let settings = RenderSettings::default() + .with_resolution(W, H) + .with_samples_per_pixel(4) + .with_max_depth(4) + .with_adaptive_sampling(4, 0.0); Renderer::new(camera, world, lights, settings) } diff --git a/crates/crust-core/tests/render_smoke.rs b/crates/crust-core/tests/render_smoke.rs index 3bdcd778..42275969 100644 --- a/crates/crust-core/tests/render_smoke.rs +++ b/crates/crust-core/tests/render_smoke.rs @@ -51,7 +51,11 @@ fn emissive_ball_scene(l: f32, w: usize, h: usize, spp: u32) -> Renderer { 0.0, 5.0, ); - let settings = RenderSettings::new(spp, 4, w, h, spp, 0.0, 0); + let settings = RenderSettings::default() + .with_resolution(w, h) + .with_samples_per_pixel(spp) + .with_max_depth(4) + .with_adaptive_sampling(spp, 0.0); Renderer::new(camera, world.commit(), LightList::new(), settings) } @@ -61,7 +65,12 @@ fn emissive_ball_scene(l: f32, w: usize, h: usize, spp: u32) -> Renderer { #[test] fn render_settings_report_what_they_were_given() { - let s = RenderSettings::new(16, 7, 320, 200, 4, 0.02, 3); + let s = RenderSettings::default() + .with_resolution(320, 200) + .with_samples_per_pixel(16) + .with_max_depth(7) + .with_adaptive_sampling(4, 0.02) + .with_frame(3); assert_eq!(s.get_dimensions(), (320, 200)); assert_eq!(s.samples_per_pixel(), 16); assert_eq!(s.max_depth(), 7); @@ -77,7 +86,11 @@ fn render_settings_report_what_they_were_given() { #[test] fn samples_per_pixel_override_is_floored_at_one() { - let s = RenderSettings::new(16, 7, 8, 8, 4, 0.02, 0); + let s = RenderSettings::default() + .with_resolution(8, 8) + .with_samples_per_pixel(16) + .with_max_depth(7) + .with_adaptive_sampling(4, 0.02); assert_eq!(s.with_samples_per_pixel(0).samples_per_pixel(), 1); assert_eq!(s.with_samples_per_pixel(9).samples_per_pixel(), 9); // Unrelated fields are untouched. @@ -86,7 +99,11 @@ fn samples_per_pixel_override_is_floored_at_one() { #[test] fn strategy_and_filter_builders_replace_their_field() { - let s = RenderSettings::new(1, 1, 8, 8, 1, 0.0, 0) + let s = RenderSettings::default() + .with_resolution(8, 8) + .with_samples_per_pixel(1) + .with_max_depth(1) + .with_adaptive_sampling(1, 0.0) .with_sampling_strategy(SamplingStrategy::LightOnly) .with_pixel_filter(PixelFilter::Mitchell { radius: 2.0 }); assert_eq!(s.sampling_strategy(), SamplingStrategy::LightOnly); @@ -100,7 +117,11 @@ fn strategy_and_filter_builders_replace_their_field() { fn guiding_builder_clamps_its_probability() { // Guiding has no getter, so the clamp is observable only through a // render completing — but the builder must at least accept edge values. - let s = RenderSettings::new(2, 2, 4, 4, 2, 0.0, 0); + let s = RenderSettings::default() + .with_resolution(4, 4) + .with_samples_per_pixel(2) + .with_max_depth(2) + .with_adaptive_sampling(2, 0.0); let _ = s.with_guiding(true, 0, 0.0); let _ = s.with_guiding(true, 100, 1.0); let off = s.with_guiding(false, 3, 0.5); @@ -203,7 +224,12 @@ fn a_different_frame_changes_the_noise_but_not_the_mean_much() { camera, b.commit(), lights, - RenderSettings::new(16, 3, w, h, 16, 0.0, frame), + RenderSettings::default() + .with_resolution(w, h) + .with_samples_per_pixel(16) + .with_max_depth(3) + .with_adaptive_sampling(16, 0.0) + .with_frame(frame), ) }; let a = mk(0).render(); @@ -306,7 +332,11 @@ fn adaptive_sampling_takes_fewer_camera_rays_on_a_flat_image() { let camera = Camera::new(Vec3A::ZERO, -Vec3A::Z, Vec3A::Y, 40.0, 1.0, 0.0, 5.0); // 64 spp allowed, minimum 4, and a zero-variance image: every pixel // stops at the first check past the minimum. - let settings = RenderSettings::new(64, 2, w, h, 4, 0.01, 0); + let settings = RenderSettings::default() + .with_resolution(w, h) + .with_samples_per_pixel(64) + .with_max_depth(2) + .with_adaptive_sampling(4, 0.01); let r = Renderer::new(camera, world.commit(), LightList::new(), settings); let (buf, stats) = r.render_with_stats(false, &|_, _| {}); assert!( @@ -344,7 +374,11 @@ fn scene_new_and_with_volumes_assemble_a_renderer() { camera, world.commit(), LightList::new(), - RenderSettings::new(1, 2, 4, 4, 1, 0.0, 0), + RenderSettings::default() + .with_resolution(4, 4) + .with_samples_per_pixel(1) + .with_max_depth(2) + .with_adaptive_sampling(1, 0.0), ) .with_volumes(Vec::new()); assert!(scene.volumes.is_empty()); @@ -600,7 +634,12 @@ fn clamp_scene(bounce_wall: bool, clamp: f32) -> Renderer { ); // Adaptive stopping off, so every setting takes the same samples and the // renders differ by the clamp alone. - let settings = RenderSettings::new(16, 6, 32, 24, 16, 0.0, 0).with_indirect_clamp(clamp); + let settings = RenderSettings::default() + .with_resolution(32, 24) + .with_samples_per_pixel(16) + .with_max_depth(6) + .with_adaptive_sampling(16, 0.0) + .with_indirect_clamp(clamp); Renderer::new(camera, world.commit(), lights, settings) } @@ -783,7 +822,11 @@ fn light_geometry_hidden_from_camera_rays_still_lights_the_scene() { camera, world.commit(), lights, - RenderSettings::new(8, 3, w, h, 8, 0.0, 0), + RenderSettings::default() + .with_resolution(w, h) + .with_samples_per_pixel(8) + .with_max_depth(3) + .with_adaptive_sampling(8, 0.0), ); let buf = r.render(); let centre = buf.get_pixel(3, 3); @@ -813,8 +856,12 @@ fn flat_adaptive_scene(l: f32, w: usize, h: usize, spp: u32, min: u32, t: f32) - Arc::new(Emissive::new(Vec3A::splat(l))), ); let camera = Camera::new(Vec3A::ZERO, -Vec3A::Z, Vec3A::Y, 40.0, 1.0, 0.0, 5.0); - let settings = - RenderSettings::new(spp, 2, w, h, min, 0.01, 0).with_adaptive_neighbour_tolerance(t); + let settings = RenderSettings::default() + .with_resolution(w, h) + .with_samples_per_pixel(spp) + .with_max_depth(2) + .with_adaptive_sampling(min, 0.01) + .with_adaptive_neighbour_tolerance(t); Renderer::new(camera, world.commit(), LightList::new(), settings) } @@ -826,7 +873,11 @@ fn adaptive_sampling_never_stops_a_pixel_that_has_seen_no_light() { let (w, h, spp) = (6, 5, 64); let world = WorldBuilder::new(); let camera = Camera::new(Vec3A::ZERO, -Vec3A::Z, Vec3A::Y, 40.0, 1.0, 0.0, 5.0); - let settings = RenderSettings::new(spp, 2, w, h, 4, 0.01, 0); + let settings = RenderSettings::default() + .with_resolution(w, h) + .with_samples_per_pixel(spp) + .with_max_depth(2) + .with_adaptive_sampling(4, 0.01); let r = Renderer::new(camera, world.commit(), LightList::new(), settings); let (buf, stats) = r.render_with_stats(false, &|_, _| {}); assert_eq!(buffer_sum(&buf, w, h), 0.0, "an empty world is black"); @@ -869,7 +920,11 @@ fn one_noisy_pixel_scene(w: usize, h: usize, spp: u32, t: f32) -> Renderer { Arc::new(Emissive::new(Vec3A::splat(400.0))), ); let camera = Camera::new(Vec3A::ZERO, -Vec3A::Z, Vec3A::Y, 40.0, 1.0, 0.0, 5.0); - let settings = RenderSettings::new(spp, 2, w, h, 4, 0.01, 0) + let settings = RenderSettings::default() + .with_resolution(w, h) + .with_samples_per_pixel(spp) + .with_max_depth(2) + .with_adaptive_sampling(4, 0.01) .with_pixel_filter(PixelFilter::BoxFilter { radius: 0.5 }) .with_adaptive_neighbour_tolerance(t); Renderer::new(camera, world.commit(), LightList::new(), settings) diff --git a/crates/crust-core/tests/resolve.rs b/crates/crust-core/tests/resolve.rs index ef5b78e7..6421f230 100644 --- a/crates/crust-core/tests/resolve.rs +++ b/crates/crust-core/tests/resolve.rs @@ -63,8 +63,7 @@ fn hit() -> HitRecord { id: 3, uv: (0.3, 0.7), }), - uv: (0.4, 0.6), - has_uv: true, + uv: Some((0.4, 0.6)), ..HitRecord::default() } } diff --git a/crates/crust-core/tests/stats.rs b/crates/crust-core/tests/stats.rs index 2b22ae20..78ee2683 100644 --- a/crates/crust-core/tests/stats.rs +++ b/crates/crust-core/tests/stats.rs @@ -112,7 +112,11 @@ fn report_sorts_the_time_view_largest_first() { #[test] fn image_counters_come_from_render_settings() { - let settings = RenderSettings::new(24, 9, 300, 200, 8, 0.05, 0); + let settings = RenderSettings::default() + .with_resolution(300, 200) + .with_samples_per_pixel(24) + .with_max_depth(9) + .with_adaptive_sampling(8, 0.05); let img: ImageCounters = (&settings).into(); assert_eq!((img.width, img.height), (300, 200)); assert_eq!(img.samples_per_pixel, 24); diff --git a/crates/crust-core/tests/usd_inline.rs b/crates/crust-core/tests/usd_inline.rs index c5030495..270caaf2 100644 --- a/crates/crust-core/tests/usd_inline.rs +++ b/crates/crust-core/tests/usd_inline.rs @@ -1971,8 +1971,10 @@ fn a_subdivided_mesh_keeps_its_uv_chart() { for (x, y) in [(1.0, 1.0), (0.5, 1.5), (1.7, 0.3), (0.1, 0.1)] { let r = Ray::new(Vec3A::new(x, y, 5.0), -Vec3A::Z).with_mask(MASK_CAMERA); let hit = scene.world.intersect(&r, 1e-3, f32::INFINITY).unwrap(); - assert!(hit.rec.has_uv, "{what}: no chart at ({x}, {y})"); - let (u, v) = hit.rec.uv; + let (u, v) = hit + .rec + .uv + .unwrap_or_else(|| panic!("{what}: no chart at ({x}, {y})")); assert!( (u - x / 2.0).abs() < 1e-4 && (v - y / 2.0).abs() < 1e-4, "{what}: ({x}, {y}) reads ({u}, {v})" @@ -2033,8 +2035,7 @@ fn every_face_varying_rule_is_read() { ); let r = Ray::new(Vec3A::new(1.001, 1.001, 5.0), -Vec3A::Z).with_mask(MASK_CAMERA); let hit = scene.world.intersect(&r, 1e-3, f32::INFINITY).unwrap(); - assert!(hit.rec.has_uv, "{rule}: no chart"); - hit.rec.uv + hit.rec.uv.unwrap_or_else(|| panic!("{rule}: no chart")) }; let rules = [ "none", @@ -2270,7 +2271,10 @@ fn non_finite_shaping_inputs_fall_back() { /// How many of the stage's lights at infinity a ray of category `mask` /// sees on escaping. fn infinite_seen(scene: &Scene, mask: crust_core::RayMask) -> usize { - scene.lights.infinite_seen_by(Vec3A::ZERO, mask).count() + scene + .lights + .infinite_indexed_seen_by(Vec3A::ZERO, mask) + .count() } fn camera_hits(scene: &Scene) -> bool { diff --git a/crates/crust-core/tests/usd_scene.rs b/crates/crust-core/tests/usd_scene.rs index 5c6d1fe3..ba1bf601 100644 --- a/crates/crust-core/tests/usd_scene.rs +++ b/crates/crust-core/tests/usd_scene.rs @@ -1583,9 +1583,9 @@ fn loads_subdivision_usda() { // The textured dome kept its chart through refinement: the top face is // charted onto the unit square, so its middle reads about (0.5, 0.5). let top = cast(textured, 0.0).rec; - assert!(top.has_uv, "the refined mesh dropped its UVs"); + assert!(top.uv.is_some(), "the refined mesh dropped its UVs"); assert!( - (top.uv.0 - 0.5).abs() < 0.05 && (top.uv.1 - 0.5).abs() < 0.05, + (top.uv.unwrap().0 - 0.5).abs() < 0.05 && (top.uv.unwrap().1 - 0.5).abs() < 0.05, "top-face middle reads {:?}", top.uv ); @@ -1680,9 +1680,8 @@ fn probe_hit() -> (crust_core::HitRecord, crust_core::Ray) { t: 1.0, front_face: true, face: None, - uv: (0.5, 0.5), + uv: Some((0.5, 0.5)), tangent: Vec3A::X, - has_uv: true, // Point-sample: this probe reports what the graph evaluates to at a // named (u, v), not what a filtered render would show there. uv_width: 0.0, @@ -1852,7 +1851,7 @@ fn face_varying_st_reaches_the_shading_point() { let probe = |x: f32, y: f32| -> Option<(f32, f32)> { let r = Ray::new(Vec3A::new(x, y, 5.0), Vec3A::new(0.0, 0.0, -1.0)).with_mask(MASK_CAMERA); let hit = scene.world.intersect(&r, 0.001, f32::INFINITY)?; - hit.rec.has_uv.then_some(hit.rec.uv) + hit.rec.uv }; let a = probe(-1.1, 1.0).expect("left quad carries no UV"); @@ -1952,11 +1951,9 @@ def Xform "W" {{ .world .intersect(&r, 0.001, f32::INFINITY) .unwrap_or_else(|| panic!("no hit at x = {x}")); - assert!( - hit.rec.has_uv, - "no chart reached the shading point at x = {x}" - ); - hit.rec.uv + hit.rec + .uv + .unwrap_or_else(|| panic!("no chart reached the shading point at x = {x}")) }; // 1. Different charts: each prim must report the coordinates it authored. @@ -2038,7 +2035,7 @@ fn untextured_geometry_carries_no_uv_table() { .intersect(&r, 0.001, f32::INFINITY) .expect("no hit in the cornell box"); assert!( - !hit.rec.has_uv, + !hit.rec.uv.is_some(), "an untextured mesh built a UV table it will never read" ); } @@ -3091,11 +3088,11 @@ def Xform "W" .world .intersect(&r, 0.001, f32::INFINITY) .expect("hits the quad"); - assert!(hit.rec.has_uv, "the perfuv chart was not read"); + assert!(hit.rec.uv.is_some(), "the perfuv chart was not read"); assert!( - (hit.rec.uv.0 - 1.5).abs() < 0.01, + (hit.rec.uv.unwrap().0 - 1.5).abs() < 0.01, "u = {}, expected ~1.5 from primvars:perfuv", - hit.rec.uv.0 + hit.rec.uv.unwrap().0 ); } @@ -3305,7 +3302,11 @@ fn skipping_stage_teardown_leaves_the_render_unchanged() { .expect("stage opens"); let geometries = scene.world.count(); // 16 spp with a minimum of 16: every pixel takes exactly 16 samples. - let settings = RenderSettings::new(16, 3, 24, 16, 16, 0.0, 0); + let settings = RenderSettings::default() + .with_resolution(24, 16) + .with_samples_per_pixel(16) + .with_max_depth(3) + .with_adaptive_sampling(16, 0.0); let buf = Renderer::new(scene.camera, scene.world, scene.lights, settings).render(); let pixels: Vec<[u32; 3]> = (0..16) .flat_map(|y| (0..24).map(move |x| (x, y))) diff --git a/crates/crust-core/tests/world_material.rs b/crates/crust-core/tests/world_material.rs index 2616db98..9abe2777 100644 --- a/crates/crust-core/tests/world_material.rs +++ b/crates/crust-core/tests/world_material.rs @@ -83,7 +83,7 @@ fn intersect_resolves_the_hit_material_and_geometry() { assert!(hit.rec.normal.abs_diff_eq(-Vec3A::Z, 1e-5)); assert!(hit.rec.front_face); assert_eq!(hit.rec.face, None, "no face table was installed"); - assert!(!hit.rec.has_uv); + assert!(hit.rec.uv.is_none()); assert_eq!(hit.rec.tangent, Vec3A::ZERO); assert!( world @@ -535,9 +535,9 @@ fn world_hits_carry_uv_and_tangent_from_the_uv_map() { let hit = world .intersect(&Ray::new(Vec3A::new(x, y, -1.0), Vec3A::Z), 1e-3, 10.0) .unwrap(); - assert!(hit.rec.has_uv); + assert!(hit.rec.uv.is_some()); assert!( - approx(hit.rec.uv.0, x, 1e-4) && approx(hit.rec.uv.1, y, 1e-4), + approx(hit.rec.uv.unwrap().0, x, 1e-4) && approx(hit.rec.uv.unwrap().1, y, 1e-4), "{:?}", hit.rec.uv ); @@ -558,15 +558,15 @@ fn a_geometry_may_carry_both_side_tables() { .intersect(&Ray::new(Vec3A::new(0.6, 0.2, -1.0), Vec3A::Z), 1e-3, 10.0) .unwrap(); assert_eq!(hit.rec.face.map(|f| f.id), Some(0)); - assert!(hit.rec.has_uv); + assert!(hit.rec.uv.is_some()); assert!(approx( hit.rec.face.expect("a face hit").uv.0, - hit.rec.uv.0, + hit.rec.uv.unwrap().0, 1e-5 )); assert!(approx( hit.rec.face.expect("a face hit").uv.1, - hit.rec.uv.1, + hit.rec.uv.unwrap().1, 1e-5 )); } @@ -1384,7 +1384,7 @@ fn instanced_meshes_get_tangents_through_their_placement() { let hit = world .intersect(&Ray::new(Vec3A::new(4.5, 0.5, -1.0), Vec3A::Z), 1e-3, 10.0) .expect("the placed quad is hit"); - assert!(hit.rec.has_uv); + assert!(hit.rec.uv.is_some()); assert!( hit.rec.tangent.abs_diff_eq(Vec3A::Y, 1e-5), "{}", @@ -1432,7 +1432,7 @@ fn forwarded_placements_sharing_a_slot_get_no_tangent() { .intersect(&Ray::new(origin, Vec3A::Z), 1e-3, 10.0) .expect("the placed quad is hit"); assert_eq!(hit.geom_id, slot, "hits report the forwarded slot"); - assert!(hit.rec.has_uv, "the slot's chart still resolves"); + assert!(hit.rec.uv.is_some(), "the slot's chart still resolves"); assert_eq!(hit.rec.tangent, Vec3A::ZERO, "no frame without a placement"); } } @@ -1462,6 +1462,6 @@ fn motion_blurred_instances_get_no_tangent() { let hit = world .intersect(&Ray::new(Vec3A::new(5.5, 0.5, -1.0), Vec3A::Z), 1e-3, 10.0) .expect("the moving quad is hit at time 0"); - assert!(hit.rec.has_uv); + assert!(hit.rec.uv.is_some()); assert_eq!(hit.rec.tangent, Vec3A::ZERO); } diff --git a/crates/crust-mtlx/src/eval.rs b/crates/crust-mtlx/src/eval.rs index eb3ae8e6..b40c53ba 100644 --- a/crates/crust-mtlx/src/eval.rs +++ b/crates/crust-mtlx/src/eval.rs @@ -95,7 +95,7 @@ pub enum Op { offset: [f32; 2], arity: u8, /// Where to look up relative to the shading point, in footprint - /// widths (see [`shifted_uv`]). Zero everywhere except the copies of + /// widths (see `shifted_uv`). Zero everywhere except the copies of /// a subgraph `heighttonormal` differentiates. shift: [f32; 2], /// The slot holding an authored `texcoord` connection, when the @@ -375,7 +375,7 @@ impl Program { /// /// - **Constant folding.** An op whose operands are all constant, and /// which reads neither the shading point nor a texture, is evaluated - /// here — by [`apply`], the function the interpreter runs, so the value + /// here — by `apply`, the function the interpreter runs, so the value /// is the one every hit would have computed, bit for bit. /// - **Constant hoisting and deduplication.** Constants move into /// [`Program::consts`], copied in with one `memcpy` per evaluation @@ -919,12 +919,12 @@ fn normal_map(encoded: Val, scale: Val, ctx: &ShadeCtx) -> Vec3A { /// Rotates a **decoded** tangent-space normal (`z` along `normal`) into world /// space, against `tangent` re-orthogonalised to `normal`. /// -/// The half of [`normal_map`] that knows nothing about MaterialX's `[0,1]` +/// The half of `normal_map` that knows nothing about MaterialX's `[0,1]` /// encoding, public so a host with its own decode — UsdPreviewSurface's /// `normal` input arrives already in `[-1,1]`, its UsdUVTexture's /// `scale`/`bias` having done the decode — rotates it identically. Returns /// `normal` unchanged when `tangent` is zero (no chart frame) or parallel to -/// it, for the reason [`normal_map`] gives. +/// it, for the reason `normal_map` gives. pub fn perturb_normal(local: Vec3A, normal: Vec3A, tangent: Vec3A) -> Vec3A { if tangent.length_squared() < 1e-20 { return normal; diff --git a/crates/crust-mtlx/src/lib.rs b/crates/crust-mtlx/src/lib.rs index 24e74448..e6fb52c5 100644 --- a/crates/crust-mtlx/src/lib.rs +++ b/crates/crust-mtlx/src/lib.rs @@ -10,15 +10,15 @@ //! //! The pipeline is three modules: //! -//! - [`parse`] — XML → a flat, name-addressable node graph. -//! - [`eval`] — that graph compiled once into a slot-indexed [`Program`], +//! - `parse` — XML → a flat, name-addressable node graph ([`Doc`]). +//! - `eval` — that graph compiled once into a slot-indexed [`Program`], //! evaluated per shading point with no name lookups and no allocation. -//! - [`bsdf`] — the closure half of the graph read as the tree MaterialX +//! - `bsdf` — the closure half of the graph read as the tree MaterialX //! defines: BSDF leaves combined by `layer` / `mix` / `add` / `multiply` //! ([`Closures`]), with the EDF terms and the interior volume beside it. //! The three surface-shader nodes (`open_pbr_surface`, `standard_surface`, //! `gltf_pbr`) expand into the tree of their MaterialX nodegraphs -//! ([`surface`]). +//! (`surface`). //! //! [`compile`] runs all three for one material node. The crate decodes *no //! pixels* and knows no colour space: both are the [`Host`]'s. An `image` @@ -42,13 +42,14 @@ //! streaming parser partly to keep it that way. #![forbid(unsafe_code)] -pub mod bsdf; -pub mod eval; -pub mod hair; -pub mod parse; -pub mod surface; +// Private modules: every public item has exactly one path, at the crate root. +mod bsdf; +mod eval; +mod hair; +mod parse; +mod surface; mod texture; -pub mod value; +mod value; pub use bsdf::{ Bsdf, Closure, Closures, DiffuseModel, Emission, Leaf, NodeId, ScatterMode, SheenMode, Slot, @@ -57,9 +58,17 @@ pub use bsdf::{ pub use eval::{ BinOp, Compiler, Op, Program, ShadeCtx, UnOp, perturb_normal, reflectivity_from_ior, }; +/// The nodedef input tables the hair nodes and the surface-shader expansions +/// are built from, transcribed from MaterialX's own `stdlib` / `pbrlib` +/// definitions (pinned against them by `tests/nodedefs.rs`). +pub use hair::{ + CHIANG_HAIR_ABSORPTION_FROM_COLOR, CHIANG_HAIR_BSDF, CHIANG_HAIR_ROUGHNESS, + DEON_HAIR_ABSORPTION_FROM_MELANIN, +}; pub use parse::{Doc, Input, MtlxError, Node, Source}; +pub use surface::{GLTF_PBR, InputDef, OPEN_PBR_SURFACE, STANDARD_SURFACE}; pub use texture::{Texture, TextureRef}; -pub use value::Val; +pub use value::{Val, arity_of, parse_literal}; /// Resolves an `image` node's `file` input — as authored, relative to the /// document — into a sampler. The second argument is the colour space the diff --git a/crates/crust-mtlx/src/parse.rs b/crates/crust-mtlx/src/parse.rs index ef4ed56d..e85d47e9 100644 --- a/crates/crust-mtlx/src/parse.rs +++ b/crates/crust-mtlx/src/parse.rs @@ -257,7 +257,7 @@ impl Doc { /// Answered whatever the input's type. Only `color3` / `color4` values — /// and the `file` of an image whose output is one — are colour-managed; /// deciding that is the caller's half (see - /// [`crate::value::is_color_type`]). + /// `is_color_type`). pub fn colorspace_of<'a>(&'a self, node: &'a Node, input: &'a Input) -> Option<&'a str> { input .colorspace diff --git a/crates/crust-mtlx/tests/graph.rs b/crates/crust-mtlx/tests/graph.rs index 5d3dcb6a..c8453819 100644 --- a/crates/crust-mtlx/tests/graph.rs +++ b/crates/crust-mtlx/tests/graph.rs @@ -252,7 +252,7 @@ fn val_from_vec3a() { #[test] fn parse_literal_handles_edge_cases() { - use crust_mtlx::value::{arity_of, parse_literal}; + use crust_mtlx::{arity_of, parse_literal}; assert!(parse_literal("", "float").is_none()); assert!(parse_literal("abc", "float").is_none()); assert!(parse_literal("1, x", "vector2").is_none()); diff --git a/crates/crust-mtlx/tests/nodedefs.rs b/crates/crust-mtlx/tests/nodedefs.rs index 70e8c0be..78f12261 100644 --- a/crates/crust-mtlx/tests/nodedefs.rs +++ b/crates/crust-mtlx/tests/nodedefs.rs @@ -5,12 +5,12 @@ //! update not carried over — would make every document that leaves an input //! unauthored shade with the wrong value, and render plausibly. -use crust_mtlx::hair::{ +use crust_mtlx::parse_literal; +use crust_mtlx::{ CHIANG_HAIR_ABSORPTION_FROM_COLOR, CHIANG_HAIR_BSDF, CHIANG_HAIR_ROUGHNESS, DEON_HAIR_ABSORPTION_FROM_MELANIN, }; -use crust_mtlx::surface::{GLTF_PBR, InputDef, OPEN_PBR_SURFACE, STANDARD_SURFACE}; -use crust_mtlx::value::parse_literal; +use crust_mtlx::{GLTF_PBR, InputDef, OPEN_PBR_SURFACE, STANDARD_SURFACE}; use std::path::Path; /// `(name, type, value)` of every input of `nodedef` in `file`, with an diff --git a/crates/crust-render/examples/maketx.rs b/crates/crust-render/examples/maketx.rs index 61f9f357..422f0c20 100644 --- a/crates/crust-render/examples/maketx.rs +++ b/crates/crust-render/examples/maketx.rs @@ -37,7 +37,7 @@ //! A `` / `` token converts the whole set, one `.tx` per tile, //! which is how the renderer expects to find them. -use crust_assets::tiled::TxFormat; +use crust_assets::TxFormat; use crust_core::ColorSpace; use std::path::{Path, PathBuf}; @@ -147,14 +147,13 @@ fn main() { } /// One tile, through the same conversion `crust-render --auto-tx` runs -/// (`crust_assets::tiled::make_tx`), written beside the source. +/// (`crust_assets::make_tx`), written beside the source. fn convert( src: &Path, space: ColorSpace, format: TxFormat, ) -> Result<(PathBuf, &'static str, u64, u64), String> { - let made = - crust_assets::tiled::make_tx_atomic(src, space, format).map_err(|e| e.to_string())?; + let made = crust_assets::make_tx_atomic(src, space, format).map_err(|e| e.to_string())?; if made.clipped { eprintln!( "warning: {} holds values above 1.0 that a TIFF backing clips — \ diff --git a/crates/crust-render/examples/mtlx_shade.rs b/crates/crust-render/examples/mtlx_shade.rs index a89774cf..0ef65b63 100644 --- a/crates/crust-render/examples/mtlx_shade.rs +++ b/crates/crust-render/examples/mtlx_shade.rs @@ -124,9 +124,8 @@ fn main() { t: 1.0, front_face: true, face: None, - uv: (u, v), + uv: Some((u, v)), tangent: Vec3A::X, - has_uv: true, // Point-sample: this probe reports what the graph evaluates to at a // named (u, v), not what a filtered render would show there. uv_width: 0.0, diff --git a/crates/crust-render/examples/tex_probe.rs b/crates/crust-render/examples/tex_probe.rs index 49c80663..f9aa6cdf 100644 --- a/crates/crust-render/examples/tex_probe.rs +++ b/crates/crust-render/examples/tex_probe.rs @@ -249,11 +249,11 @@ fn probe_ptex(path: &str) { println!("texels sampled {count}"); println!("mean raw (as stored, 0..1) = {mean:.4}"); println!( - " -> if file is sRGB, linear mean = {:.4}", + " -> if file is gamma 2.2, linear mean = {:.4}", mean.powf(2.2) ); println!( - " -> if file is linear, sRGB mean = {:.4}", + " -> if file is linear, gamma 2.2 mean = {:.4}", mean.powf(1.0 / 2.2) ); print!("decile histogram:"); diff --git a/crates/crust-rt/examples/ray_throughput.rs b/crates/crust-rt/examples/ray_throughput.rs index 667c0c11..03e37332 100644 --- a/crates/crust-rt/examples/ray_throughput.rs +++ b/crates/crust-rt/examples/ray_throughput.rs @@ -171,7 +171,6 @@ static LAYOUT: std::sync::OnceLock = std::sync::OnceLock::new(); fn commit(b: SceneBuilder) -> Scene { b.commit_with(CommitOptions { layout: *LAYOUT.get().unwrap_or(&PacketLayout::Auto), - ..Default::default() }) } diff --git a/crates/crust-rt/src/bvh/build.rs b/crates/crust-rt/src/bvh/build.rs index 35dcf394..90f0325d 100644 --- a/crates/crust-rt/src/bvh/build.rs +++ b/crates/crust-rt/src/bvh/build.rs @@ -303,7 +303,6 @@ pub(super) fn build_subtree( mut refs: Vec, depth: usize, root_area: f32, - packet_sah: bool, ) -> Subtree { let bbox = union_all(&refs); let count = refs.len(); @@ -322,10 +321,7 @@ pub(super) fn build_subtree( // By packet cost a small all-triangle range whose object split does // not pay is a leaf before any spatial split is weighed: chopping it // would only duplicate references into packets that are half empty. - // (The per-triangle rule never leafed on this path; keeping that is - // what makes `packet_sah = false` the behaviour it replaces.) - if packet_sah - && all_triangles + if all_triangles && count <= MAX_LEAF && let Some(o) = &object && !splitting_pays(o, &bbox, count, true) @@ -392,18 +388,14 @@ pub(super) fn build_subtree( if left.is_empty() || right.is_empty() { let mut refs: Vec = left; refs.extend(right); - return object_partition_or_leaf( - prims, refs, bbox, object, depth, root_area, packet_sah, - ); + return object_partition_or_leaf(prims, refs, bbox, object, depth, root_area); } (left, right) } else { match object { Some(o) => { // Leaf when splitting costs more than intersecting through. - if count <= MAX_LEAF - && !splitting_pays(&o, &bbox, count, all_triangles && packet_sah) - { + if count <= MAX_LEAF && !splitting_pays(&o, &bbox, count, all_triangles) { return leaf(bbox, &refs); } // By value: the parent's buffer is freed here rather than @@ -422,13 +414,13 @@ pub(super) fn build_subtree( let parallel = left_refs.len().max(right_refs.len()) > PARALLEL_THRESHOLD; let (l, r) = if parallel { rayon::join( - || build_subtree(prims, left_refs, depth + 1, root_area, packet_sah), - || build_subtree(prims, right_refs, depth + 1, root_area, packet_sah), + || build_subtree(prims, left_refs, depth + 1, root_area), + || build_subtree(prims, right_refs, depth + 1, root_area), ) } else { ( - build_subtree(prims, left_refs, depth + 1, root_area, packet_sah), - build_subtree(prims, right_refs, depth + 1, root_area, packet_sah), + build_subtree(prims, left_refs, depth + 1, root_area), + build_subtree(prims, right_refs, depth + 1, root_area), ) }; merge(bbox, l, r) @@ -436,10 +428,11 @@ pub(super) fn build_subtree( /// Whether the object split beats keeping the range as one leaf. /// -/// Per triangle (`packets` false, the rule before `CommitOptions::packet_sah`): -/// the split's `Σ area · count` against the leaf's `area · count`, with no -/// node cost, so a range only stays a leaf when its children are nearly as -/// large as it is. +/// Per primitive (`packets` false, any range holding a non-triangle): the +/// split's `Σ area · count` against the leaf's `area · count`, with no node +/// cost, so a range only stays a leaf when its children are nearly as large +/// as it is. All-triangle ranges used this rule too until packet-sized +/// leaves (the retired `CRUST_BVH_PACKET_SAH` A/B). /// /// Per packet (`packets` true, all-triangle ranges only): a leaf of `n` /// triangles costs `ceil(n / 4)` SIMD rounds, the split the same per side @@ -462,7 +455,6 @@ fn splitting_pays(o: &ObjSplit, bbox: &AABB, count: usize, packets: bool) -> boo /// The non-spatial tail of `build_subtree`, reused by the degenerate-chop /// fallback: object-partition when possible, else leaf. -#[allow(clippy::too_many_arguments)] fn object_partition_or_leaf( prims: &Primitives, refs: Vec, @@ -470,7 +462,6 @@ fn object_partition_or_leaf( object: Option, depth: usize, root_area: f32, - packet_sah: bool, ) -> Subtree { let count = refs.len(); match object { @@ -481,8 +472,8 @@ fn object_partition_or_leaf( all.extend(r); return leaf(bbox, &all); } - let left = build_subtree(prims, l, depth + 1, root_area, packet_sah); - let right = build_subtree(prims, r, depth + 1, root_area, packet_sah); + let left = build_subtree(prims, l, depth + 1, root_area); + let right = build_subtree(prims, r, depth + 1, root_area); merge(bbox, left, right) } _ => leaf(bbox, &refs), diff --git a/crates/crust-rt/src/bvh/lane_width.rs b/crates/crust-rt/src/bvh/lane_width.rs index e7c3bec4..27f7b05e 100644 --- a/crates/crust-rt/src/bvh/lane_width.rs +++ b/crates/crust-rt/src/bvh/lane_width.rs @@ -39,7 +39,7 @@ fn uv_sphere_prims(segs: usize, rings: usize) -> Primitives { /// Records what 8-wide packets would buy, now that leaves can hold two. /// -/// Until the packet-aware leaf cost (`CommitOptions::packet_sah`) no leaf on +/// Until the packet-aware leaf cost (packet-sized leaves) no leaf on /// a dense mesh held more than four triangles, so an 8-wide leaf intersector /// (AVX2) would have run exactly as many vector rounds as the 4-wide one with /// half its lanes idle — the equality this test used to pin. Leaves of five to @@ -51,7 +51,7 @@ fn uv_sphere_prims(segs: usize, rings: usize) -> Primitives { /// slower) are unchanged by it. #[test] fn eight_wide_packets_would_save_at_most_the_two_packet_leaves() { - let bvh = Bvh::new(uv_sphere_prims(80, 40), Layout::Gathered, true); + let bvh = Bvh::new(uv_sphere_prims(80, 40), Layout::Gathered); let per_leaf: Vec = bvh .leaves .iter() diff --git a/crates/crust-rt/src/bvh/mod.rs b/crates/crust-rt/src/bvh/mod.rs index cdf09ca5..17d882a3 100644 --- a/crates/crust-rt/src/bvh/mod.rs +++ b/crates/crust-rt/src/bvh/mod.rs @@ -474,7 +474,7 @@ impl Candidate { } impl Bvh { - pub(crate) fn new(input: Primitives, layout: Layout, packet_sah: bool) -> Self { + pub(crate) fn new(input: Primitives, layout: Layout) -> Self { // One reference per primitive, in input order (the order decides // ties, so it is part of the build's determinism); degenerate // records get none. @@ -486,7 +486,7 @@ impl Bvh { (Vec::new(), LeafData::default(), None) } else { let root_bbox = union_all(&refs); - let subtree = build_subtree(&input, refs, 0, surface_area(&root_bbox), packet_sah); + let subtree = build_subtree(&input, refs, 0, surface_area(&root_bbox)); let (wide, collected) = collapse(&subtree.nodes, &subtree.indices, &input, layout); (wide, collected, Some(root_bbox)) }; diff --git a/crates/crust-rt/src/bvh/tests.rs b/crates/crust-rt/src/bvh/tests.rs index 51ccb407..b7f24a0f 100644 --- a/crates/crust-rt/src/bvh/tests.rs +++ b/crates/crust-rt/src/bvh/tests.rs @@ -203,7 +203,7 @@ fn linear_scan(prims: &Primitives, ray: &Ray, t_min: f32, t_max: f32) -> Option< } fn assert_matches_linear(objects: impl Fn() -> Primitives) { - let bvh = Bvh::new(objects(), Layout::Gathered, true); + let bvh = Bvh::new(objects(), Layout::Gathered); let reference = objects(); let origins = [ @@ -264,7 +264,7 @@ fn spatial_splits_match_linear_scan() { /// two places now: packed SIMD lanes and the scalar `indices` list. #[test] fn spatial_splits_duplicate_references() { - let bvh = Bvh::new(diagonal_shards(64), Layout::Gathered, true); + let bvh = Bvh::new(diagonal_shards(64), Layout::Gathered); let refs = bvh.leaf_ref_count(); assert!( refs > bvh.prim_count(), @@ -277,7 +277,7 @@ fn spatial_splits_duplicate_references() { /// triangle must be packed rather than left on the scalar path. #[test] fn triangles_are_packed_into_simd_lanes() { - let bvh = Bvh::new(diagonal_shards(64), Layout::Gathered, true); + let bvh = Bvh::new(diagonal_shards(64), Layout::Gathered); assert!( !bvh.packets.is_empty(), "no packets built for a triangle scene" @@ -289,14 +289,14 @@ fn triangles_are_packed_into_simd_lanes() { ); // Spheres are not packable and must stay on the scalar path. - let bvh = Bvh::new(sphere_grid(4), Layout::Gathered, true); + let bvh = Bvh::new(sphere_grid(4), Layout::Gathered); assert!(bvh.packets.is_empty(), "spheres must not be packed"); assert_eq!(bvh.indices.len(), bvh.leaf_ref_count()); // Mixed leaves must place each primitive on exactly one path. let mut mixed = diagonal_shards(16); mixed.append(sphere_grid(2)); - let bvh = Bvh::new(mixed, Layout::Gathered, true); + let bvh = Bvh::new(mixed, Layout::Gathered); assert!(!bvh.packets.is_empty() && !bvh.indices.is_empty()); assert!(bvh.leaf_ref_count() >= bvh.prim_count()); } @@ -327,31 +327,16 @@ fn packet_aware_leaves_keep_overlapping_triangles_together() { MASK_ALL, ); } - let packed = Bvh::new( - Primitives { - tris: prims.tris.clone(), - vertices: prims.vertices.clone(), - normals: Vec::new(), - geoms: prims.geoms.clone(), - ..Default::default() - }, - Layout::Gathered, - true, - ); - let per_tri = Bvh::new(prims, Layout::Gathered, false); + let packed = Bvh::new(prims, Layout::Gathered); assert_eq!(packed.leaves.len(), 1, "one leaf of two packets"); assert_eq!(packed.packets.len(), 2); - assert!( - per_tri.leaves.len() > 1, - "the per-triangle cost splits them" - ); } /// Packet lanes must average close to 4 on a dense mesh — a packing /// that mostly emitted 1-lane packets would be SIMD in name only. #[test] fn packets_are_well_filled() { - let bvh = Bvh::new(diagonal_shards(256), Layout::Gathered, true); + let bvh = Bvh::new(diagonal_shards(256), Layout::Gathered); let lanes: u32 = bvh .packets .iter() @@ -367,7 +352,7 @@ fn packets_are_well_filled() { /// `hit_any` must agree with `hit(..).is_some()` for every ray and range. #[test] fn hit_any_matches_hit() { - let bvh = Bvh::new(sphere_grid(4), Layout::Gathered, true); + let bvh = Bvh::new(sphere_grid(4), Layout::Gathered); let origins = [ Vec3A::new(-5.0, 4.5, 4.5), Vec3A::new(20.0, 3.0, 3.0), @@ -458,8 +443,8 @@ fn wide8_node_is_four_cache_lines() { /// always produces byte-identical topology. #[test] fn build_is_deterministic() { - let a = Bvh::new(sphere_grid(6), Layout::Gathered, true); - let b = Bvh::new(sphere_grid(6), Layout::Gathered, true); + let a = Bvh::new(sphere_grid(6), Layout::Gathered); + let b = Bvh::new(sphere_grid(6), Layout::Gathered); assert_eq!(a.wide.len(), b.wide.len()); assert_eq!(a.indices, b.indices); assert_eq!(a.packets.len(), b.packets.len()); @@ -480,7 +465,7 @@ fn build_is_deterministic() { /// fewer wide nodes than a binary tree would need. #[test] fn collapse_widens_the_tree() { - let bvh = Bvh::new(sphere_grid(6), Layout::Gathered, true); // 216 prims + let bvh = Bvh::new(sphere_grid(6), Layout::Gathered); // 216 prims let n_leaf_slots: usize = bvh .wide .iter() @@ -513,7 +498,7 @@ fn collapsed_tables_hold_no_spare_capacity() { .map(|i| PrimRef::new(prims.bbox(i).expect("no degenerate records here"), i)) .collect(); let root = union_all(&refs); - let subtree = build_subtree(&prims, refs, 0, surface_area(&root), true); + let subtree = build_subtree(&prims, refs, 0, surface_area(&root)); let (wide, collected) = collapse(&subtree.nodes, &subtree.indices, &prims, Layout::Gathered); assert_eq!(wide.capacity(), wide.len()); @@ -525,7 +510,7 @@ fn collapsed_tables_hold_no_spare_capacity() { #[test] fn empty_bvh_misses() { - let bvh = Bvh::new(Primitives::default(), Layout::Gathered, true); + let bvh = Bvh::new(Primitives::default(), Layout::Gathered); let ray = Ray::new(Vec3A::ZERO, Vec3A::X); assert!(bvh.hit(&ray, 0.001, f32::INFINITY).is_none()); assert!(bvh.bounds().is_none()); @@ -533,7 +518,7 @@ fn empty_bvh_misses() { #[test] fn bounds_cover_all_prims() { - let bvh = Bvh::new(sphere_grid(3), Layout::Gathered, true); + let bvh = Bvh::new(sphere_grid(3), Layout::Gathered); let bbox = bvh.bounds().expect("grid is fully bounded"); assert!(bbox.minimum.cmple(Vec3A::splat(-0.5)).all()); assert!(bbox.maximum.cmpge(Vec3A::splat(6.5)).all()); diff --git a/crates/crust-rt/src/scene.rs b/crates/crust-rt/src/scene.rs index 6c37ba32..c15e7176 100644 --- a/crates/crust-rt/src/scene.rs +++ b/crates/crust-rt/src/scene.rs @@ -52,23 +52,12 @@ pub enum PacketLayout { pub struct CommitOptions { /// The triangle packet layout. pub layout: PacketLayout, - /// Size all-triangle leaves by packet tests rather than triangle tests: - /// a range of four or fewer triangles is one SIMD round whatever its - /// count, so the SAH leaf decision charges `ceil(n / 4)` per side plus - /// one node test for the split, and a range of five to eight triangles - /// whose children overlap stays one leaf of two full packets instead of - /// splitting into two half-empty ones. `false` is the per-triangle - /// leaf cost before this option existed. Either way the build is - /// deterministic; the two trees differ in shape, so their renders can - /// differ on exact-tie hits only. - pub packet_sah: bool, } impl Default for CommitOptions { fn default() -> Self { CommitOptions { layout: PacketLayout::Auto, - packet_sah: true, } } } @@ -661,7 +650,7 @@ impl SceneBuilder { PacketLayout::Indexed => crate::bvh::Layout::Indexed, }; Scene { - bvh: Bvh::new(input, layout, options.packet_sah), + bvh: Bvh::new(input, layout), n_geoms, has_motion, max_hit_id, diff --git a/crates/crust-rt/tests/kernel.rs b/crates/crust-rt/tests/kernel.rs index 7380253f..dee614a2 100644 --- a/crates/crust-rt/tests/kernel.rs +++ b/crates/crust-rt/tests/kernel.rs @@ -1611,10 +1611,7 @@ fn packet_layouts_are_bit_identical() { gt.clone(), )); b.attach(sphere(Vec3A::new(0.0, 0.0, 2.5), 0.4)); - b.commit_with(crust_rt::CommitOptions { - layout, - ..Default::default() - }) + b.commit_with(crust_rt::CommitOptions { layout }) }; let gathered = build(PacketLayout::Gathered); let indexed = build(PacketLayout::Indexed); diff --git a/docs/architecture.md b/docs/architecture.md index a7febda7..5faf28be 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -224,8 +224,7 @@ probe that needs another setting builds a `Config` and passes it | `CRUST_DISPLACE` | on | `usd_import/materials.rs` | `0`: no material yields a displacement — every mesh imports undisplaced, a `none` mesh as its faceted cage (bit-identical to the stage with its displacement inputs removed) | | `CRUST_ADAPTIVE_PER_FACE` | on | `usd_import/mesh.rs` (`mesh_source`) | In adaptive subdivision only. `0`: refine each unshared subdivision mesh to one level instead of tessellating it per face at its edges' own rates | | `CRUST_ADAPTIVE_FRUSTUM` | on | `usd_import/adaptive.rs` (`Frustum`) | In adaptive subdivision only. `0`: rate geometry outside the camera's view by distance like the rest, instead of splitting each of its edges once | -| `CRUST_BVH_PACKET_SAH` | on | `lib.rs` (`commit_options`) → every kernel `commit` | `0`: the per-triangle SAH leaf cost before packet-sized leaves (five to eight overlapping triangles split into two half-empty packets). Not bit-identical: the trees differ in shape, so exact-tie hits can differ; proven noise by the 1/√N check in the design record | -| `CRUST_TRI_PACKETS` | `auto` (= `gathered`) | `lib.rs` (`packet_layout`) → every kernel `commit` | `gathered`: 192-byte vertex-carrying packets (the layout before indexed packets); `indexed`: 92-byte index packets, a quarter fewer kernel bytes per triangle for 13–30% slower traversal (8–9% on the Moana island at level 1, where a 3–4% faster import makes the whole run faster) — the opt-in for a scene that otherwise does not fit. Bit-identical | +| `CRUST_TRI_PACKETS` | `auto` (= `gathered`) | `lib.rs` (`commit_options`) → every kernel `commit` | `gathered`: 192-byte vertex-carrying packets (the layout before indexed packets); `indexed`: 92-byte index packets, a quarter fewer kernel bytes per triangle for 13–30% slower traversal (8–9% on the Moana island at level 1, where a 3–4% faster import makes the whole run faster) — the opt-in for a scene that otherwise does not fit. Bit-identical | | `CRUST_MTLX_OPT` | on | `material/materialx.rs` | `0`: skip constant folding / hoisting / pruning (bit-identical) | | `CRUST_SHADER_JIT` | on | `material/materialx.rs` | `0`: interpret MaterialX programs instead of JIT (bit-identical) | | `CRUST_RAY_CONES` | on | `tracer/path.rs` | `0`: zero every texture footprint (finest mip always) | diff --git a/docs/light_sampling.md b/docs/light_sampling.md index 27fcf454..db55f63a 100644 --- a/docs/light_sampling.md +++ b/docs/light_sampling.md @@ -407,7 +407,7 @@ contiguous slice of that dimension, as under uniform picking. **Tests.** - Probabilities and pick frequencies match the rule, with a dark light never picked. -- `find_by_geom`, `iter` and `pick` agree. +- `find_index_by_geom_at`, `iter` and `pick` agree. - The uniform density is the historical division. - End to end, a bright and a dim sphere, a sun and a dome estimate the same radiance under power and uniform selection, with MIS and with light sampling @@ -694,7 +694,8 @@ none of the rest: power, distance and orientation cannot see a wall. quantile bounds, sized for about 16 receivers per occupied cell. - **Per-cell tables.** A cell with at least 2 receivers and any light seen gets `p = 0.7 · E/ΣE + 0.3 / n_live`; every other point uses the power table. -- **MIS.** `LightList::pick_at` / `pmf_at` / `find_by_geom_at` / `iter_at` serve +- **MIS.** `LightList::pick_index_at` / `pmf_at` / `find_index_by_geom_at` / + `infinite_at` serve both MIS sides from the same stored `f32`, at the vertex NEE sampled from. The bounce side reads it at `prev.pos`, which is that vertex. - **Unbiased.** The defensive share gives every light that emits at least @@ -1374,7 +1375,7 @@ tracking. - `LightList::select_by` builds a CDF (not an alias table, to keep the pick dimension's stratification) over `Light::power`. - `LightList::density(pdf, pmf)` replaces `pdf / n_lights` at all four sites. -- `find_by_geom` is an O(1) `geom_id → index` map. +- `index_of_geom` (behind `find_index_by_geom_at`) is an O(1) `geom_id → index` map. - Infinite lights get their uniform share, and half the finite lights' rays are spread evenly, because the pure form measured worse (§3.8). diff --git a/docs/simd.md b/docs/simd.md index 8348dcf2..60c331a3 100644 --- a/docs/simd.md +++ b/docs/simd.md @@ -71,7 +71,7 @@ four triangles, because the SAH was free to split anything above `MIN_LEAF_PACKED`, and a `Tri8` packet would have run the *same number* of vector rounds with half the lanes idle. The retune the old test warned about has happened — `compact-triangle-storage` sizes all-triangle leaves by packet -tests (`CommitOptions::packet_sah`, `CRUST_BVH_PACKET_SAH`), so five to eight +tests (the `CRUST_BVH_PACKET_SAH` A/B, since retired), so five to eight overlapping triangles now stay one leaf of two full packets instead of two half-empty ones. On the same sphere mesh 607 of 1 316 leaves hold two packets, and an 8-wide packet would merge those pairs: 1 923 → 1 316 rounds, diff --git a/openspec/specs/cli/design.md b/openspec/specs/cli/design.md index 3e2452bd..72c4bfad 100644 --- a/openspec/specs/cli/design.md +++ b/openspec/specs/cli/design.md @@ -69,11 +69,6 @@ cargo run --release -- -i samples/curves.usda --stats python3 scripts/gen_subdiv_stress.py /tmp/subdiv_stress.usda CRUST_TRI_PACKETS=gathered target/release/crust-render -i /tmp/subdiv_stress.usda --stats -l error CRUST_TRI_PACKETS=indexed target/release/crust-render -i /tmp/subdiv_stress.usda --stats -l error -# ...and the leaf rule: CRUST_BVH_PACKET_SAH=0 is the per-triangle SAH leaf cost -# that left production scenes' packet lanes 48% full; on (the default) sizes -# all-triangle leaves by packet rounds. Different tree, so compare its images -# by the 1/sqrt(N) rule, not bit for bit. -CRUST_BVH_PACKET_SAH=0 target/release/crust-render -i samples/cornellbox.usda --stats -l error # ...and inside the render: Trace vs EvalBsdfs vs Texture vs SurfaceLighting, # flat / by category / by execution tree. Costs render time (~15-20%, printed # with the report), so never take a Render time from a --profile run. diff --git a/openspec/specs/cli/spec.md b/openspec/specs/cli/spec.md index fe92c5ae..c84f3dd7 100644 --- a/openspec/specs/cli/spec.md +++ b/openspec/specs/cli/spec.md @@ -120,13 +120,12 @@ triangle, and the kernel bytes per resident triangle. `triangle packets (gathered)` or `(indexed)`, `lanes filled` as a percentage and `bytes per triangle`, and the rows sum to `kernel memory` -### Requirement: Two geometry-layout switches +### Requirement: A geometry-layout switch `CRUST_TRI_PACKETS` (`gathered` | `indexed` | `auto`, default `auto`) SHALL force the -packet layout on every tree, and `CRUST_BVH_PACKET_SAH` (boolean, default on) SHALL -select the packet-aware leaf cost; both SHALL be parsed once into `Config`, warn once -on a bad value, and be listed in `docs/architecture.md` with the behaviour before this -change as their off side (`gathered`, off). +packet layout on every tree; it SHALL be parsed once into `Config`, warn once on a bad +value, and be listed in `docs/architecture.md` with the behaviour before indexed +packets as its off side (`gathered`). #### Scenario: A bad value diff --git a/openspec/specs/intersection-kernel/design.md b/openspec/specs/intersection-kernel/design.md index 05143352..80bb4d85 100644 --- a/openspec/specs/intersection-kernel/design.md +++ b/openspec/specs/intersection-kernel/design.md @@ -140,8 +140,9 @@ 4.38 → 3.82 GiB; every sample, the Kitchen_set pair and the island and ALab frames bit-identical, in both packet layouts; callgrind −0.7% (cornellbox) and −0.6% (nested_instancing) instructions overall, `Scene::occluded` −4.8% / −3.4%. - **Leaves are sized by packet rounds** (`CommitOptions::packet_sah`, the - `CRUST_BVH_PACKET_SAH` switch, default on). The SAH leaf decision used to charge one + **Leaves are sized by packet rounds** (once the `CRUST_BVH_PACKET_SAH` A/B, retired + when the measurements below settled it; the per-triangle cost survives only for + ranges holding a non-triangle). The SAH leaf decision used to charge one unit per triangle with no node cost, so a range of five to eight overlapping triangles always split into two half-empty packets — on ALab and the Moana island packet lanes were 48% full, and every half-empty packet is 192 resident bytes, a @@ -279,7 +280,7 @@ model). Current practice says target AVX2 instead, so the reasons not to are recorded: merely *enabling* AVX2 codegen is worth 2–4% (LLVM cannot widen a 4-lane algorithm), and 8-wide leaf packets used to buy exactly nothing because no leaf held more than 4 - triangles; with packet-sized leaves (`CommitOptions::packet_sah`, below) a leaf holds + triangles; with packet-sized leaves (below) a leaf holds up to two packets, and an 8-wide packet would merge those pairs — 31.6% fewer rounds on the test sphere mesh, measured and bounded by `eight_wide_packets_would_save_at_most_the_two_packet_leaves` — which changes the diff --git a/openspec/specs/intersection-kernel/spec.md b/openspec/specs/intersection-kernel/spec.md index c7d0d477..8d4fa478 100644 --- a/openspec/specs/intersection-kernel/spec.md +++ b/openspec/specs/intersection-kernel/spec.md @@ -153,20 +153,16 @@ behaviour before this change. ### Requirement: Leaves are sized for packets -With `CRUST_BVH_PACKET_SAH` on (the default), the builder SHALL charge an all-triangle -range one intersection cost per packet of four rather than per triangle when deciding -whether to make it a leaf, so that leaves fill their packet lanes. With it off, the -builder SHALL use the per-primitive cost it used before this change. Either setting -SHALL build deterministically. The two settings MAY differ in which of two triangles -at exactly the same hit distance is reported, and in nothing else; the difference -between their renders SHALL fall as 1/√N with the sample count. +The builder SHALL charge an all-triangle range one intersection cost per packet of four +rather than per triangle when deciding whether to make it a leaf, so that leaves fill +their packet lanes, and SHALL build deterministically. A range holding any other +primitive SHALL keep the per-primitive cost. #### Scenario: Overlapping triangles -- **WHEN** six overlapping triangles that no split separates well are committed with the - switch on -- **THEN** they form one leaf of two packets, the second with two inactive lanes; with - it off they form more than one leaf +- **WHEN** six overlapping triangles that no split separates well are committed +- **THEN** they form one leaf of two packets, the second with two inactive lanes, where + the per-triangle cost would have split them #### Scenario: Lane fill is reported diff --git a/openspec/specs/lighting/design.md b/openspec/specs/lighting/design.md index aa6314e0..923f8ae3 100644 --- a/openspec/specs/lighting/design.md +++ b/openspec/specs/lighting/design.md @@ -94,7 +94,7 @@ visible) when the crust attribute is not authored, an authored `crust:rayMask` wins outright, and shadow/indirect rays always see it) — the `AreaLight` records the geometry's `geom_id`, which is how the integrator - attributes a bounce-hit emissive surface to its light (`LightList::find_by_geom`). + attributes a bounce-hit emissive surface to its light (`LightList::find_index_by_geom_at`). **NEE samples one light per vertex**, picked by the `LightList`'s selection, `crust:lightSelection` / `--light-selection`, built in `Renderer::new` from the settings. @@ -121,8 +121,8 @@ traverse the whole BVH. So at equal time it is ~3.6× on ALab's direct lighting and only ~1.1× on its full image, which is mostly indirect. Four details are load-bearing: - - **MIS.** Both sides go through `LightList::pick_at` / `pmf_at` / - `find_by_geom_at` / `iter_at`, keyed by the vertex NEE sampled from. The + - **MIS.** Both sides go through `LightList::pick_index_at` / `pmf_at` / + `find_index_by_geom_at` / `infinite_at`, keyed by the vertex NEE sampled from. The bounce side passes `prev.pos`. Route a new pmf read through the `*_at` form or emission is double-counted. - **Robust grid bounds.** The grid spans the receivers' 2–98% quantiles. @@ -152,7 +152,7 @@ - time within noise. The light strategy's MIS density is `light.pdf · pmf`, computed by - `LightList::density` on **both** sides. `pick`, `find_by_geom` and `iter` all hand + `LightList::density` on **both** sides. `pick`, `find_index_by_geom_at` and `iter` all hand back the same `pmf` for the same light. Under uniform, `density` is the historical division `pdf / n`, not `pdf · (1/n)`, which rounds differently when n is not a power of two; that is what keeps the A/B exact. A light with `pmf = 0` keeps its bounce diff --git a/openspec/specs/rendering/design.md b/openspec/specs/rendering/design.md index 48094b90..478035f9 100644 --- a/openspec/specs/rendering/design.md +++ b/openspec/specs/rendering/design.md @@ -195,7 +195,12 @@ consumed as ordinary dependencies: the training budget is not discarded. Delta/transmissive materials (`Material::eval` → `None`) and untrained regions fall back to pure BSDF sampling. The NEE weight competes against the same mixture pdf — keep the two sides - consistent or emission gets double-counted. + consistent or emission gets double-counted. The quadtree descent draws a fresh pair + per level from the `K_GUIDE` domain's `rng()`; it used to hash the guide seed into a + hand-rolled PCG32 outside `openqmc`. Switching streams changed `cornellbox_guided`'s + noise only: against the old stream, relmse 3.1e-2 / 2.1e-2 / 7.6e-3 at 16 / 64 / + 256 spp (`--indirect-clamp 0`), no plateau, and both stand exactly as far from an + unguided 256-spp reference. The training passes double as a **guiding efficiency estimate** (Li et al. 2026, "Path Guiding in Disney's Zootopia 2"): efficiency `E = 1/(wall-clock cost × MRSE)`, comparing the first pass (field untrained → effectively unguided) against the last @@ -398,7 +403,7 @@ domain, and hands the root to `trace_path`. Each path vertex derives `path.new_d and each sampling event a further keyed sub-domain (`K_NEE`, `K_BSDF`, `K_GUIDE`, `K_PHASE`, …, keys defined atop `tracer/path.rs`); materials draw one 4D block from the `SobolSampler` domain they are handed. Unbounded/incidental draws — Russian roulette, volume delta-tracking, -carried-medium free flight — use `draw_rnd` or a `pcg::Rng` seeded from a domain +carried-medium free flight, the guide's quadtree descent — use `draw_rnd` or a `pcg::Rng` seeded from a domain (`domain.rng()`), matching OpenQMC's `drawSample` vs `drawRnd` split. Tests that just need randomness use `openqmc::pcg::Rng`. diff --git a/site/content/docs/reference/environment-variables.md b/site/content/docs/reference/environment-variables.md index 4bb159c0..94ec9d92 100644 --- a/site/content/docs/reference/environment-variables.md +++ b/site/content/docs/reference/environment-variables.md @@ -71,7 +71,6 @@ then the default is used. A typo never stops a render, so read the warnings. | [`CRUST_DISPLACE`](#crust-displace) | on | USD import | | [`CRUST_ADAPTIVE_PER_FACE`](#crust-adaptive-per-face) | on | USD import | | [`CRUST_ADAPTIVE_FRUSTUM`](#crust-adaptive-frustum) | on | USD import | -| [`CRUST_BVH_PACKET_SAH`](#crust-bvh-packet-sah) | on | ray tracing | | [`CRUST_TRI_PACKETS`](#crust-tri-packets) | `auto` | ray tracing | | [`CRUST_MTLX_OPT`](#crust-mtlx-opt) | on | shading | | [`CRUST_SHADER_JIT`](#crust-shader-jit) | on | shading | @@ -154,16 +153,6 @@ reflections and shadows keep their detail, at the cost of memory. ## Ray tracing -### CRUST_BVH_PACKET_SAH - -Boolean, default **on**. - -On, the BVH builder sizes its leaves by how many SIMD packets of triangles they hold. `0` -uses the older per-triangle leaf cost. - -The trees have different shapes, so ties between exactly equal hits can resolve -differently. The difference is noise, not bias. - ### CRUST_TRI_PACKETS Keyword, default **`auto`**. From 8328036bb9c78052b14e058aed277ab7fe123980 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 5 Oct 2026 08:51:24 +0000 Subject: [PATCH 4/5] Split the long functions and files - scene/subdiv.rs (2,406 lines) -> subdiv/{mod,topology,uniform,adaptive, normals,tests}.rs; tessellate_adaptive (588 lines) becomes a driver over Cage, EdgeRating and Emitted, with tessellate_selected and tessellate_unselected for its two phases. - stats.rs: the 687-line Display::fmt dispatches to write_scene / write_rays / write_textures / write_ptex / write_phases. - closure::prepare: one prepare_* builder per BSDF kind over LeafInputs. - usd_import/mesh.rs: mesh_source takes MeshNeeds (5 parameters, not 8) and delegates to effective_scheme, vtx_boundary, SharpnessArrays, refinable_chart and MeshSource::{tessellated,subdivided}. - usd_import load_scene: ValidOptions, traverse_stage, resolve_camera and record_geometry_counters taken out; ImageCounters reuses its From impl. - trace_path takes a PathContext (6 parameters, not 12). Kept to the inlined function only: handing &PathContext to the out-of-line helpers cost cornellbox 0.08% (by value, 0.23%); as committed it runs 0.03% fewer instructions than before (callgrind, 2 spp). - crust-mtlx: eval.rs -> eval/{mod,apply,compile,tests}.rs; surface.rs -> surface/{mod,open_pbr,standard_surface,gltf_pbr}.rs. - crust-rt scene.rs -> scene/{mod,tests}.rs; commit_with sizes the arrays in sized_primitives and expands each geometry through Expansion::add. Fixes a latent bug found on the way: a skipped geometry (invalid disk or cylinder, instance of an empty scene) pushed no GeomTable, but the tables are indexed by geom_id, so every later triangle mesh read its neighbour's bases and the last one indexed past the end (panic); pinned by a_skipped_geometry_keeps_its_table_slot. - crust-render: logging setup -> logging.rs, RenderProducts -> products.rs, the traversal-stats table -> crust_core::traversal_report; main() loses load_scene / apply_overrides / select_products (212 lines). Every sample bit-identical at 16 spp; workspace tests pass. MaterialX shading costs +0.055% instructions on materialx_surfaces (prepare_*). Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_019nwge6NCTPhuk1VPuZucRF --- CLAUDE.md | 3 +- crates/crust-core/src/lib.rs | 2 + crates/crust-core/src/material/closure/mod.rs | 547 ++-- crates/crust-core/src/scene/subdiv.rs | 2406 ----------------- .../crust-core/src/scene/subdiv/adaptive.rs | 904 +++++++ crates/crust-core/src/scene/subdiv/mod.rs | 172 ++ crates/crust-core/src/scene/subdiv/normals.rs | 66 + crates/crust-core/src/scene/subdiv/tests.rs | 1069 ++++++++ .../crust-core/src/scene/subdiv/topology.rs | 166 ++ crates/crust-core/src/scene/subdiv/uniform.rs | 243 ++ .../src/scene/usd_import/instancing.rs | 7 +- .../crust-core/src/scene/usd_import/mesh.rs | 436 +-- crates/crust-core/src/scene/usd_import/mod.rs | 359 +-- crates/crust-core/src/stats.rs | 166 +- crates/crust-core/src/tracer/mod.rs | 19 +- crates/crust-core/src/tracer/path.rs | 57 +- crates/crust-mtlx/src/eval.rs | 2185 --------------- crates/crust-mtlx/src/eval/apply.rs | 517 ++++ crates/crust-mtlx/src/eval/compile.rs | 680 +++++ crates/crust-mtlx/src/eval/mod.rs | 483 ++++ crates/crust-mtlx/src/eval/tests.rs | 525 ++++ crates/crust-mtlx/src/surface.rs | 1362 ---------- crates/crust-mtlx/src/surface/gltf_pbr.rs | 259 ++ crates/crust-mtlx/src/surface/mod.rs | 525 ++++ crates/crust-mtlx/src/surface/open_pbr.rs | 332 +++ .../src/surface/standard_surface.rs | 251 ++ crates/crust-render/src/logging.rs | 268 ++ crates/crust-render/src/main.rs | 923 +------ crates/crust-render/src/products.rs | 378 +++ crates/crust-rt/src/scene.rs | 1833 ------------- crates/crust-rt/src/scene/mod.rs | 908 +++++++ crates/crust-rt/src/scene/tests.rs | 991 +++++++ docs/architecture.md | 13 +- docs/color_management.md | 2 +- docs/rust_leverage.md | 2 +- docs/shading_performance.md | 2 +- openspec/specs/image-output/design.md | 3 +- openspec/specs/image-output/spec.md | 3 +- openspec/specs/materials/design.md | 11 +- openspec/specs/usd-scene-import/design.md | 2 +- 40 files changed, 9861 insertions(+), 9219 deletions(-) delete mode 100644 crates/crust-core/src/scene/subdiv.rs create mode 100644 crates/crust-core/src/scene/subdiv/adaptive.rs create mode 100644 crates/crust-core/src/scene/subdiv/mod.rs create mode 100644 crates/crust-core/src/scene/subdiv/normals.rs create mode 100644 crates/crust-core/src/scene/subdiv/tests.rs create mode 100644 crates/crust-core/src/scene/subdiv/topology.rs create mode 100644 crates/crust-core/src/scene/subdiv/uniform.rs delete mode 100644 crates/crust-mtlx/src/eval.rs create mode 100644 crates/crust-mtlx/src/eval/apply.rs create mode 100644 crates/crust-mtlx/src/eval/compile.rs create mode 100644 crates/crust-mtlx/src/eval/mod.rs create mode 100644 crates/crust-mtlx/src/eval/tests.rs delete mode 100644 crates/crust-mtlx/src/surface.rs create mode 100644 crates/crust-mtlx/src/surface/gltf_pbr.rs create mode 100644 crates/crust-mtlx/src/surface/mod.rs create mode 100644 crates/crust-mtlx/src/surface/open_pbr.rs create mode 100644 crates/crust-mtlx/src/surface/standard_surface.rs create mode 100644 crates/crust-render/src/logging.rs create mode 100644 crates/crust-render/src/products.rs delete mode 100644 crates/crust-rt/src/scene.rs create mode 100644 crates/crust-rt/src/scene/mod.rs create mode 100644 crates/crust-rt/src/scene/tests.rs diff --git a/CLAUDE.md b/CLAUDE.md index 11bde3fe..a077cce9 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -87,7 +87,8 @@ Seven crates under `crates/` (ownership table in `docs/architecture.md`): deps), `crust-jit` (Cranelift JIT for `crust-mtlx` programs, feature `jit`), `crust-core` (the engine library: import, integrator, materials, lights, volumes, guiding, stats), `crust-assets` (every file decoder and texture cache, behind `crust_core::AssetLoader`), -`crust-render` (the CLI; `main.rs` only writes images) and `utils` (stateless math: +`crust-render` (the CLI; it only drives a render and writes images: `main.rs`, +`products.rs`, `logging.rs`) and `utils` (stateless math: warps, MIS heuristics, the one Rec.709 `luminance`). `openqmc-rs` (all sampling) and `opensubdiv-rs` / `ptex-rs` are external. Import from `crust_core::` roots; `lib.rs` re-exports the public surface. diff --git a/crates/crust-core/src/lib.rs b/crates/crust-core/src/lib.rs index 00cf589a..223c0634 100644 --- a/crates/crust-core/src/lib.rs +++ b/crates/crust-core/src/lib.rs @@ -97,6 +97,8 @@ pub use ray::{ pub use rt_world::{FaceMap, FanSlice, SubFace, UvMap, World, WorldBuilder, WorldHit, tangent_of}; pub use scene::Scene; pub use scene::{AssetLoader, NoAssets, UsdImportOptions}; +#[cfg(feature = "traversal-stats")] +pub use stats::traversal_report; pub use stats::{ DisplacementCounters, ImageCounters, MemorySample, Phase, PrimitiveCounts, PtexCacheStats, RayStats, RenderStats, SceneCounters, SubdivisionCounters, TextureCacheStats, diff --git a/crates/crust-core/src/material/closure/mod.rs b/crates/crust-core/src/material/closure/mod.rs index f30e3a4a..107f9581 100644 --- a/crates/crust-core/src/material/closure/mod.rs +++ b/crates/crust-core/src/material/closure/mod.rs @@ -1024,7 +1024,6 @@ fn alphas(v: Val) -> (f32, f32) { /// leaf, whose throughput is 0. fn prepare(leaf: &crust_mtlx::Leaf, iface: Interface, w: &Walk<'_>) -> (Prepared, Option) { let s = |i: u32| w.slots[i as usize]; - let rgb = |i: u32| sanitize(s(i).rgb()); let n = leaf .normal .map(|i| s(i).rgb()) @@ -1043,133 +1042,23 @@ fn prepare(leaf: &crust_mtlx::Leaf, iface: Interface, w: &Walk<'_>) -> (Prepared } let v = frame.to_local(w.v_world); let nv = v.z.clamp(1e-6, 1.0); - let film = |tf: &Option| { - tf.map(|t| (s(t.thickness).x().max(0.0), s(t.ior).x().max(1.0))) - .filter(|(d, _)| *d > 0.0) - }; + let cx = LeafInputs { w, iface, nv, v }; let (lobe, select, throughput) = match &leaf.bsdf { Bsdf::Diffuse { model, color, roughness, - } => { - let color = rgb(*color); - let r = s(*roughness).x().clamp(0.0, 1.0); - let albedo = match model { - DiffuseModel::Eon => mx::eon_dir_albedo(nv, r, color), - DiffuseModel::OrenNayar => color * mx::oren_nayar_dir_albedo(nv, r), - DiffuseModel::Burley => color * mx::burley_dir_albedo(nv, r), - }; - ( - Lobe::Diffuse { - model: *model, - color, - roughness: r, - }, - w.luma.of(albedo).max(0.02), - None, - ) - } + } => prepare_diffuse(&cx, model, color, roughness), Bsdf::Subsurface { color, radius, anisotropy, - } => { - let color = rgb(*color).min(Vec3A::ONE); - // `radius` is a vector3; a float broadcasts. - let r = s(*radius); - let radius = sanitize(if r.arity >= 3 { - r.rgb() - } else { - Vec3A::splat(r.x()) - }); - let lobe = if radius.max_element() > 0.0 { - let anisotropy = s(*anisotropy).x(); - Lobe::Subsurface { - color, - radius, - anisotropy: if anisotropy.is_finite() { - anisotropy.clamp(-0.99, 0.99) - } else { - 0.0 - }, - ior: iface.ior, - alpha: iface.alpha, - } - } else { - // A zero mean free path exits where it entered: a diffuse in - // the subsurface colour, which is also what MaterialX's GLSL - // renders it as. - Lobe::Diffuse { - model: DiffuseModel::OrenNayar, - color, - roughness: 0.0, - } - }; - (lobe, w.luma.of(color).max(0.02), None) - } - Bsdf::Hair { - tint_r, - tint_tt, - tint_trt, - ior, - roughness_r, - roughness_tt, - roughness_trt, - cuticle_angle, - absorption, - } => { - // A `vector2` (variance, scale); a `float` broadcasts. - let pair = |i: u32| { - let r = s(i); - if r.arity >= 2 { - (r.v[0], r.v[1]) - } else { - (r.x(), r.x()) - } - }; - let hair = hair::Hair::new( - &hair::HairParams { - tint: [rgb(*tint_r), rgb(*tint_tt), rgb(*tint_trt)], - ior: s(*ior).x(), - roughness: [ - pair(*roughness_r), - pair(*roughness_tt), - pair(*roughness_trt), - ], - cuticle_angle: s(*cuticle_angle).x(), - absorption: rgb(*absorption), - }, - v.normalize_or_zero(), - |c| w.luma.of(c), - ); - let albedo = hair.albedo(); - // Over a base, a fibre passes on what it does not scatter. - (Lobe::Hair(hair), w.luma.of(albedo).max(0.02), Some(albedo)) - } - Bsdf::Translucent { color } => { - let color = rgb(*color); - ( - Lobe::Translucent { color }, - w.luma.of(color).max(0.02), - None, - ) - } + } => prepare_subsurface(&cx, color, radius, anisotropy), + Bsdf::Hair { .. } => prepare_hair(&cx, &leaf.bsdf), + Bsdf::Translucent { color } => prepare_translucent(&cx, color), Bsdf::Sheen { color, roughness, .. - } => { - let color = rgb(*color); - let r = s(*roughness).x().clamp(0.0, 1.0); - let e = mx::imageworks_sheen_dir_albedo(nv, r); - ( - Lobe::Sheen { - color, - roughness: r, - }, - (w.luma.of(color) * e).max(0.02), - Some(Vec3A::splat(e)), - ) - } + } => prepare_sheen(&cx, color, roughness), Bsdf::Dielectric { tint, ior, @@ -1177,123 +1066,14 @@ fn prepare(leaf: &crust_mtlx::Leaf, iface: Interface, w: &Walk<'_>) -> (Prepared mode, thin_film, .. - } => { - let ior = s(*ior).x(); - let ior = if ior.is_finite() && ior > 0.0 { - ior - } else { - 1.5 - }; - let (ax, ay) = alphas(s(*roughness)); - let tint = rgb(*tint); - // In the ray-facing frame: entering the interior from the front, - // leaving it from the back. - let eta = if w.rec.front_face { ior } else { 1.0 / ior }; - let fresnel = Fresnel { - model: FresnelModel::Dielectric { ior: eta }, - thin_film: film(thin_film), - }; - let avg = mx::average_alpha(ax, ay); - // MaterialX's throughput for every dielectric mode: `1 − E_R·w`, - // the reflection albedo alone. BSDL's table where it applies (no - // film); MaterialX's Fresnel-weighted fit with a film. - let e_r = if fresnel.thin_film.is_some() { - let f = fresnel.eval(nv); - fresnel.dir_albedo(nv, avg) * mx::ggx_energy_compensation(nv, avg, f) - } else { - Vec3A::splat(1.0 - dielectric_refl_filter(nv, avg.sqrt(), ior)) - }; - let e = w.luma.of(e_r); - let select = match mode { - ScatterMode::R => e, - ScatterMode::T => (1.0 - e) * w.luma.of(tint), - ScatterMode::RT => e + (1.0 - e) * w.luma.of(tint), - }; - ( - Lobe::Specular { - fresnel, - tint, - ax, - ay, - mode: *mode, - eta, - thin_walled: w.thin_walled, - }, - select.max(0.02), - Some(e_r), - ) - } + } => prepare_dielectric(&cx, tint, ior, roughness, mode, thin_film), Bsdf::Conductor { ior, extinction, roughness, thin_film, - } => { - let (ax, ay) = alphas(s(*roughness)); - let fresnel = Fresnel { - model: FresnelModel::Conductor { - n: rgb(*ior), - k: rgb(*extinction), - }, - thin_film: film(thin_film), - }; - let avg = mx::average_alpha(ax, ay); - let e = fresnel.dir_albedo(nv, avg) - * mx::ggx_energy_compensation(nv, avg, fresnel.eval(nv)); - ( - Lobe::Specular { - fresnel, - tint: Vec3A::ONE, - ax, - ay, - mode: ScatterMode::R, - eta: 1.0, - thin_walled: false, - }, - w.luma.of(e).max(0.02), - // MaterialX: a conductor is opaque. - None, - ) - } - Bsdf::Schlick { - color0, - color82, - color90, - exponent, - roughness, - mode, - thin_film, - } => { - let (ax, ay) = alphas(s(*roughness)); - let f0 = rgb(*color0); - let fresnel = Fresnel { - model: FresnelModel::Schlick { - f0, - f82: rgb(*color82), - f90: rgb(*color90), - exponent: s(*exponent).x().max(0.0), - }, - thin_film: film(thin_film), - }; - let avg = mx::average_alpha(ax, ay); - let e = fresnel.dir_albedo(nv, avg) - * mx::ggx_energy_compensation(nv, avg, fresnel.eval(nv)); - let e_avg = (e.x + e.y + e.z) / 3.0; - let eta = mx::f0_to_ior(Vec3A::splat((f0.x + f0.y + f0.z) / 3.0)).x; - ( - Lobe::Specular { - fresnel, - tint: Vec3A::ONE, - ax, - ay, - mode: *mode, - eta: if w.rec.front_face { eta } else { 1.0 / eta }, - thin_walled: w.thin_walled, - }, - e_avg.max(0.02), - Some(Vec3A::splat(e_avg)), - ) - } + } => prepare_conductor(&cx, ior, extinction, roughness, thin_film), + Bsdf::Schlick { .. } => prepare_schlick(&cx, &leaf.bsdf), }; ( Prepared { @@ -1309,6 +1089,315 @@ fn prepare(leaf: &crust_mtlx::Leaf, iface: Interface, w: &Walk<'_>) -> (Prepared ) } +/// A leaf's lobe, its selection weight (floored at 0.02), and the directional +/// albedo its throughput is `1 − E·weight` of (`None`: opaque) — what each +/// `prepare_*` builds. +type Built = (Lobe, f32, Option); + +/// What every `prepare_*` reads: the walk's slots and hit, the interface a +/// subsurface leaf is entered through, and the view direction in the leaf's +/// frame with its clamped cosine. +struct LeafInputs<'a> { + w: &'a Walk<'a>, + iface: Interface, + nv: f32, + v: Vec3A, +} + +impl LeafInputs<'_> { + fn s(&self, i: u32) -> Val { + self.w.slots[i as usize] + } + + fn rgb(&self, i: u32) -> Vec3A { + sanitize(self.s(i).rgb()) + } + + /// A thin film's `(thickness, ior)`, when it has any thickness. + fn film(&self, tf: &Option) -> Option<(f32, f32)> { + tf.map(|t| (self.s(t.thickness).x().max(0.0), self.s(t.ior).x().max(1.0))) + .filter(|(d, _)| *d > 0.0) + } +} + +/// [`prepare`] for a `Diffuse` leaf. +#[inline(always)] +fn prepare_diffuse( + cx: &LeafInputs<'_>, + model: &DiffuseModel, + color: &u32, + roughness: &u32, +) -> Built { + let color = cx.rgb(*color); + let r = cx.s(*roughness).x().clamp(0.0, 1.0); + let albedo = match model { + DiffuseModel::Eon => mx::eon_dir_albedo(cx.nv, r, color), + DiffuseModel::OrenNayar => color * mx::oren_nayar_dir_albedo(cx.nv, r), + DiffuseModel::Burley => color * mx::burley_dir_albedo(cx.nv, r), + }; + ( + Lobe::Diffuse { + model: *model, + color, + roughness: r, + }, + cx.w.luma.of(albedo).max(0.02), + None, + ) +} + +/// [`prepare`] for a `Subsurface` leaf. +#[inline(always)] +fn prepare_subsurface(cx: &LeafInputs<'_>, color: &u32, radius: &u32, anisotropy: &u32) -> Built { + let color = cx.rgb(*color).min(Vec3A::ONE); + // `radius` is a vector3; a float broadcasts. + let r = cx.s(*radius); + let radius = sanitize(if r.arity >= 3 { + r.rgb() + } else { + Vec3A::splat(r.x()) + }); + let lobe = if radius.max_element() > 0.0 { + let anisotropy = cx.s(*anisotropy).x(); + Lobe::Subsurface { + color, + radius, + anisotropy: if anisotropy.is_finite() { + anisotropy.clamp(-0.99, 0.99) + } else { + 0.0 + }, + ior: cx.iface.ior, + alpha: cx.iface.alpha, + } + } else { + // A zero mean free path exits where it entered: a diffuse in + // the subsurface colour, which is also what MaterialX's GLSL + // renders it as. + Lobe::Diffuse { + model: DiffuseModel::OrenNayar, + color, + roughness: 0.0, + } + }; + (lobe, cx.w.luma.of(color).max(0.02), None) +} + +/// [`prepare`] for a `Hair` leaf. +#[inline(always)] +fn prepare_hair(cx: &LeafInputs<'_>, bsdf: &Bsdf) -> Built { + let Bsdf::Hair { + tint_r, + tint_tt, + tint_trt, + ior, + roughness_r, + roughness_tt, + roughness_trt, + cuticle_angle, + absorption, + } = bsdf + else { + unreachable!("prepare_hair is handed a Hair leaf") + }; + // A `vector2` (variance, scale); a `float` broadcasts. + let pair = |i: u32| { + let r = cx.s(i); + if r.arity >= 2 { + (r.v[0], r.v[1]) + } else { + (r.x(), r.x()) + } + }; + let hair = hair::Hair::new( + &hair::HairParams { + tint: [cx.rgb(*tint_r), cx.rgb(*tint_tt), cx.rgb(*tint_trt)], + ior: cx.s(*ior).x(), + roughness: [ + pair(*roughness_r), + pair(*roughness_tt), + pair(*roughness_trt), + ], + cuticle_angle: cx.s(*cuticle_angle).x(), + absorption: cx.rgb(*absorption), + }, + cx.v.normalize_or_zero(), + |c| cx.w.luma.of(c), + ); + let albedo = hair.albedo(); + // Over a base, a fibre passes on what it does not scatter. + ( + Lobe::Hair(hair), + cx.w.luma.of(albedo).max(0.02), + Some(albedo), + ) +} + +/// [`prepare`] for a `Translucent` leaf. +#[inline(always)] +fn prepare_translucent(cx: &LeafInputs<'_>, color: &u32) -> Built { + let color = cx.rgb(*color); + ( + Lobe::Translucent { color }, + cx.w.luma.of(color).max(0.02), + None, + ) +} + +/// [`prepare`] for a `Sheen` leaf. +#[inline(always)] +fn prepare_sheen(cx: &LeafInputs<'_>, color: &u32, roughness: &u32) -> Built { + let color = cx.rgb(*color); + let r = cx.s(*roughness).x().clamp(0.0, 1.0); + let e = mx::imageworks_sheen_dir_albedo(cx.nv, r); + ( + Lobe::Sheen { + color, + roughness: r, + }, + (cx.w.luma.of(color) * e).max(0.02), + Some(Vec3A::splat(e)), + ) +} + +/// [`prepare`] for a `Dielectric` leaf. +#[inline(always)] +fn prepare_dielectric( + cx: &LeafInputs<'_>, + tint: &u32, + ior: &u32, + roughness: &u32, + mode: &ScatterMode, + thin_film: &Option, +) -> Built { + let ior = cx.s(*ior).x(); + let ior = if ior.is_finite() && ior > 0.0 { + ior + } else { + 1.5 + }; + let (ax, ay) = alphas(cx.s(*roughness)); + let tint = cx.rgb(*tint); + // In the ray-facing frame: entering the interior from the front, + // leaving it from the back. + let eta = if cx.w.rec.front_face { ior } else { 1.0 / ior }; + let fresnel = Fresnel { + model: FresnelModel::Dielectric { ior: eta }, + thin_film: cx.film(thin_film), + }; + let avg = mx::average_alpha(ax, ay); + // MaterialX's throughput for every dielectric mode: `1 − E_R·w`, + // the reflection albedo alone. BSDL's table where it applies (no + // film); MaterialX's Fresnel-weighted fit with a film. + let e_r = if fresnel.thin_film.is_some() { + let f = fresnel.eval(cx.nv); + fresnel.dir_albedo(cx.nv, avg) * mx::ggx_energy_compensation(cx.nv, avg, f) + } else { + Vec3A::splat(1.0 - dielectric_refl_filter(cx.nv, avg.sqrt(), ior)) + }; + let e = cx.w.luma.of(e_r); + let select = match mode { + ScatterMode::R => e, + ScatterMode::T => (1.0 - e) * cx.w.luma.of(tint), + ScatterMode::RT => e + (1.0 - e) * cx.w.luma.of(tint), + }; + ( + Lobe::Specular { + fresnel, + tint, + ax, + ay, + mode: *mode, + eta, + thin_walled: cx.w.thin_walled, + }, + select.max(0.02), + Some(e_r), + ) +} + +/// [`prepare`] for a `Conductor` leaf. +#[inline(always)] +fn prepare_conductor( + cx: &LeafInputs<'_>, + ior: &u32, + extinction: &u32, + roughness: &u32, + thin_film: &Option, +) -> Built { + let (ax, ay) = alphas(cx.s(*roughness)); + let fresnel = Fresnel { + model: FresnelModel::Conductor { + n: cx.rgb(*ior), + k: cx.rgb(*extinction), + }, + thin_film: cx.film(thin_film), + }; + let avg = mx::average_alpha(ax, ay); + let e = fresnel.dir_albedo(cx.nv, avg) + * mx::ggx_energy_compensation(cx.nv, avg, fresnel.eval(cx.nv)); + ( + Lobe::Specular { + fresnel, + tint: Vec3A::ONE, + ax, + ay, + mode: ScatterMode::R, + eta: 1.0, + thin_walled: false, + }, + cx.w.luma.of(e).max(0.02), + // MaterialX: a conductor is opaque. + None, + ) +} + +/// [`prepare`] for a `Schlick` leaf. +#[inline(always)] +fn prepare_schlick(cx: &LeafInputs<'_>, bsdf: &Bsdf) -> Built { + let Bsdf::Schlick { + color0, + color82, + color90, + exponent, + roughness, + mode, + thin_film, + } = bsdf + else { + unreachable!("prepare_schlick is handed a Schlick leaf") + }; + let (ax, ay) = alphas(cx.s(*roughness)); + let f0 = cx.rgb(*color0); + let fresnel = Fresnel { + model: FresnelModel::Schlick { + f0, + f82: cx.rgb(*color82), + f90: cx.rgb(*color90), + exponent: cx.s(*exponent).x().max(0.0), + }, + thin_film: cx.film(thin_film), + }; + let avg = mx::average_alpha(ax, ay); + let e = fresnel.dir_albedo(cx.nv, avg) + * mx::ggx_energy_compensation(cx.nv, avg, fresnel.eval(cx.nv)); + let e_avg = (e.x + e.y + e.z) / 3.0; + let eta = mx::f0_to_ior(Vec3A::splat((f0.x + f0.y + f0.z) / 3.0)).x; + ( + Lobe::Specular { + fresnel, + tint: Vec3A::ONE, + ax, + ay, + mode: *mode, + eta: if cx.w.rec.front_face { eta } else { 1.0 / eta }, + thin_walled: cx.w.thin_walled, + }, + e_avg.max(0.02), + Some(Vec3A::splat(e_avg)), + ) +} + /// BSDL's dielectric reflection filter `1 − E_R(cosθo)` at perceptual /// roughness `r` (`α = r²`) and IOR `ior`, interpolated as BSDL's /// `TabulatedEnergyCurve` does: bilinear in (IOR, roughness), piecewise diff --git a/crates/crust-core/src/scene/subdiv.rs b/crates/crust-core/src/scene/subdiv.rs deleted file mode 100644 index 002206cc..00000000 --- a/crates/crust-core/src/scene/subdiv.rs +++ /dev/null @@ -1,2406 +0,0 @@ -//! Subdivision-surface refinement for USD meshes, via the pure-Rust -//! [`opensubdiv-rs`] port of OpenSubdiv's Far/Sdc layers. -//! -//! The importer hands this module a base cage (points + faceVertexCounts + -//! faceVertexIndices, exactly as authored) and gets back a uniformly refined -//! mesh whose positions sit **on the limit surface** and whose vertices carry -//! smooth shading normals. Everything downstream — triangulation, interning, -//! instancing, baking — then treats the refined mesh like any other polygon -//! mesh. -//! -//! Ptex face ids index the *base cage*, so when the caller needs per-face -//! texturing the refinement also reports, per refined face, which cage face -//! it descends from and where its corners sit inside that face's unit square -//! ([`SubdivFaces`]). The sub-face UVs come from a synthetic face-varying -//! channel (each cage face owns four values at the Ptex corners, so every -//! edge of the channel is a face-varying boundary and it refines bilinearly -//! under every [`FVarLinearInterpolation`] rule but `None`, which is -//! refined apart) — the channel *is* the parameterization, so the smoothing -//! rules must never touch it. -//! -//! The authored texture chart ([`UvChannel`]) is refined with the surface: -//! a `faceVarying` chart as a real face-varying channel under the mesh's -//! `faceVaryingLinearInterpolation`, a `vertex` chart like the points. Both -//! are snapped to the limit, so a texel stays on the limit-surface point its -//! vertex was snapped to. -//! -//! [`opensubdiv-rs`]: https://github.com/doubleailes/OpenSubdiv-rs -//! [`FVarLinearInterpolation`]: sdc::FVarLinearInterpolation - -use glam::Vec3A; -use opensubdiv_rs::far::{ - FVarChannelDescriptor, PrimvarRefiner, TopologyDescriptor, TopologyRefinerFactory, - UniformOptions, -}; -use opensubdiv_rs::sdc; -use openusd::gf::Vec3f; -use std::fmt; - -/// The subdivision schemes the importer maps from `subdivisionScheme`. -#[derive(Clone, Copy, PartialEq, Eq, Debug)] -pub(crate) enum SubdivScheme { - CatmullClark, - Bilinear, - /// Loop subdivision refines triangles into triangles; the factory rejects - /// any non-triangular face, so the caller pre-checks the cage. - Loop, -} - -/// Everything `subdivide` needs beyond the cage arrays, borrowed straight -/// from the authored attributes. -#[derive(Clone, Copy)] -pub(crate) struct SubdivRequest<'a> { - pub scheme: SubdivScheme, - /// Uniform refinement depth, `>= 1` (level 0 never reaches this module). - pub level: u32, - pub boundary: sdc::VtxBoundaryInterpolation, - /// USD authors creases as runs of vertices: run `i` spans - /// `crease_lengths[i]` consecutive entries of `crease_indices` and - /// describes `crease_lengths[i] - 1` edges. - pub crease_indices: &'a [i32], - pub crease_lengths: &'a [i32], - /// One sharpness per run *or* one per edge — both are legal USD. - pub crease_sharpnesses: &'a [f32], - pub corner_indices: &'a [i32], - pub corner_sharpnesses: &'a [f32], - /// Build [`SubdivFaces`] (only wanted when the material samples a - /// per-face texture). - pub want_face_uvs: bool, - /// The authored texture chart to refine, when the material reads one. - pub uvs: Option>, -} - -/// An authored texture-coordinate primvar, as the importer read it. The -/// caller checks it with [`UvChannel::is_well_formed`] first — a refiner -/// cannot skip a bad value the way a triangle lookup can. -#[derive(Clone, Copy)] -pub(crate) struct UvChannel<'a> { - pub values: &'a [[f32; 2]], - /// The primvar's `:indices` into `values`, or `None` for direct - /// addressing. - pub indices: Option<&'a [i32]>, - /// `faceVarying` (one entry per face-vertex) rather than `vertex` (one - /// per point). - pub face_varying: bool, - /// The mesh's `faceVaryingLinearInterpolation`; only a `faceVarying` - /// chart reads it. - pub linear: sdc::FVarLinearInterpolation, -} - -impl UvChannel<'_> { - /// Whether every entry the cage addresses resolves to a value: - /// `n_entries` is the face-vertex count for a `faceVarying` chart, the - /// point count for a `vertex` one. - pub(crate) fn is_well_formed(&self, n_entries: usize) -> bool { - match self.indices { - Some(idx) => { - idx.len() >= n_entries - && idx[..n_entries] - .iter() - .all(|&i| i >= 0 && (i as usize) < self.values.len()) - } - None => self.values.len() >= n_entries, - } - } - - /// The value entry `i` (a face-vertex or a point) addresses. - fn value_index(&self, i: usize) -> usize { - match self.indices { - Some(idx) => idx[i] as usize, - None => i, - } - } -} - -/// The refined chart, in the shape the importer's `UvSource` holds. -pub(crate) struct RefinedUvs { - pub values: Vec<[f32; 2]>, - /// Per refined face-vertex, into `values` — `Some` iff `face_varying`. - pub indices: Option>, - pub face_varying: bool, -} - -/// Per refined face: the base-cage face it descends from and its corner UVs -/// inside that face's unit square (Ptex convention: `v0=(0,0) v1=(1,0) -/// v2=(1,1) v3=(0,1)`). -pub(crate) struct SubdivFaces { - /// `None` for a face with no Ptex-addressable ancestor (its cage face - /// was not a quad — Ptex subfaces are out of scope, matching the - /// unsubdivided importer's treatment of n-gons). - pub base_face: Vec>, - /// Corner UVs of the refined quad, in `face_vertices` order. - pub corner_uvs: Vec<[[f32; 2]; 4]>, -} - -/// A refined mesh in the same array shapes `triangulate` consumes. -pub(crate) struct SubdividedMesh { - /// Limit-surface positions (uniform refinement, then limit snap). - pub points: Vec, - /// All 4s for Catmull-Clark/Bilinear, all 3s for Loop. - pub counts: Vec, - pub indices: Vec, - /// Smooth per-vertex shading normals, parallel to `points`, unpadded. - pub normals: Vec<[f32; 3]>, - /// `Some` iff the request asked for face UVs. - pub faces: Option, - /// `Some` iff the request carried a texture chart. - pub uvs: Option, -} - -#[derive(Debug)] -pub(crate) enum SubdivError { - /// The cage failed validation before it reached the refiner — unlike - /// `triangulate`, a topology refiner cannot skip a malformed face, so the - /// whole mesh degrades to its cage. - BadTopology(String), - Refine(opensubdiv_rs::far::Error), -} - -impl fmt::Display for SubdivError { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - match self { - SubdivError::BadTopology(why) => write!(f, "{why}"), - SubdivError::Refine(e) => write!(f, "{e}"), - } - } -} - -/// Uniformly refines the cage to `req.level` and snaps the result to the -/// limit surface. See the module docs for the shape of the answer. -pub(crate) fn subdivide( - points: &[Vec3f], - counts: &[i32], - indices: &[i32], - req: &SubdivRequest, -) -> Result { - // `none` is the one face-varying rule that smooths face-varying corners, - // which would slide the Ptex channel sharing its refiner off its - // sub-faces. Refine the face table on its own, chartless, and the rest - // without it — twice the refinement, for a material that reads Ptex *and* - // a `none` chart. (Pinned by `ptex_channel_is_invariant_under_every_fvar_rule`.) - if req.want_face_uvs - && req - .uvs - .is_some_and(|c| c.face_varying && c.linear == sdc::FVarLinearInterpolation::None) - { - let faces = subdivide( - points, - counts, - indices, - &SubdivRequest { uvs: None, ..*req }, - )? - .faces; - let mut out = subdivide( - points, - counts, - indices, - &SubdivRequest { - want_face_uvs: false, - ..*req - }, - )?; - out.faces = faces; - return Ok(out); - } - let (counts_us, indices_u32) = validate_cage(points.len(), counts, indices)?; - let (crease_pairs, crease_weights) = expand_crease_runs( - req.crease_indices, - req.crease_lengths, - req.crease_sharpnesses, - )?; - let corners = validate_corners(req.corner_indices, req.corner_sharpnesses)?; - - let scheme = match req.scheme { - SubdivScheme::CatmullClark => sdc::SchemeType::Catmark, - SubdivScheme::Bilinear => sdc::SchemeType::Bilinear, - SubdivScheme::Loop => sdc::SchemeType::Loop, - }; - // The face-varying rule is one per refiner, not per channel, so the - // authored chart's rule wins. The synthetic Ptex channel must refine - // bilinearly, and does under every rule but `none` (handled above) — - // each of its values is private to one face, so every one of its edges - // is a face-varying boundary, and its data is affine. Without a chart, - // `All`. - let chart = req.uvs.as_ref().filter(|c| c.face_varying); - let options = sdc::Options::default() - .with_vtx_boundary_interpolation(req.boundary) - .with_fvar_linear_interpolation( - chart.map_or(sdc::FVarLinearInterpolation::All, |c| c.linear), - ); - - // Loop cannot refine quads at all, and the Ptex channel only makes sense - // for the quad-split schemes. - let want_uvs = req.want_face_uvs && req.scheme != SubdivScheme::Loop; - let (fvar_uvs, fvar_indices) = if want_uvs { - ptex_fvar_channel(&counts_us) - } else { - (Vec::new(), Vec::new()) - }; - // A face-varying chart's per-face-vertex indices are its channel's - // topology; its values seed the refinement. - let chart_indices: Vec = match chart { - Some(c) => (0..indices.len()) - .map(|fv| c.value_index(fv) as u32) - .collect(), - None => Vec::new(), - }; - let mut channels = Vec::with_capacity(2); - if want_uvs { - channels.push(FVarChannelDescriptor::new(fvar_uvs.len(), &fvar_indices)); - } - let chart_channel = chart.map(|c| { - channels.push(FVarChannelDescriptor::new(c.values.len(), &chart_indices)); - channels.len() - 1 - }); - - let mut descriptor = TopologyDescriptor::new(points.len(), &counts_us, &indices_u32) - .with_creases(&crease_pairs, &crease_weights) - .with_corners(&corners.0, &corners.1); - if !channels.is_empty() { - descriptor = descriptor.with_fvar_channels(&channels); - } - - let mut refiner = - TopologyRefinerFactory::create(descriptor, scheme, options).map_err(SubdivError::Refine)?; - let level = req.level as usize; - refiner.refine_uniform(UniformOptions::new(level)); - - // Positions: interpolate level by level, then snap the last level to the - // limit surface. (For Bilinear the limit is the refined mesh itself; - // limit_level handles that uniformly.) - let primvar = PrimvarRefiner::new(&refiner); - let mut verts: Vec<[f32; 3]> = points.iter().map(|p| [p.x, p.y, p.z]).collect(); - for l in 1..=level { - let mut refined = vec![[0.0f32; 3]; refiner.level(l).num_vertices()]; - primvar.interpolate(l, &verts, &mut refined); - verts = refined; - } - let mut limit = vec![[0.0f32; 3]; verts.len()]; - primvar.limit(&verts, &mut limit); - - // Topology of the last level, back in the importer's array shapes. - let last = refiner.level(level); - let n_faces = last.num_faces(); - let mut out_counts = Vec::with_capacity(n_faces); - let mut out_indices = Vec::with_capacity(last.num_face_vertices_total()); - for f in 0..n_faces { - let fv = last.face_vertices(f); - out_counts.push(fv.len() as i32); - out_indices.extend(fv.iter().map(|&v| v as i32)); - } - - let faces = want_uvs.then(|| { - // Base-cage face per refined face: compose the one-step - // child-to-parent maps from the last refinement down to level 0. - let mut base_face: Vec = (0..n_faces as u32).collect(); - for l in (1..=level).rev() { - let refinement = refiner.refinement(l); - for f in &mut base_face { - *f = refinement.child_face_parent_face(*f as usize); - } - } - let base_face: Vec> = base_face - .into_iter() - .map(|f| (counts[f as usize] == 4).then_some(f)) - .collect(); - - // Sub-face corner UVs: refine the synthetic channel the same way the - // positions were refined, then read each face's four values. - let mut uvs = fvar_uvs.clone(); - for l in 1..=level { - let mut refined = vec![[0.0f32; 2]; refiner.level(l).num_fvar_values(0)]; - primvar.interpolate_face_varying(l, 0, &uvs, &mut refined); - uvs = refined; - } - let corner_uvs = (0..n_faces) - .map(|f| { - let fv = last.face_fvar_values(f, 0); - debug_assert_eq!(fv.len(), 4, "quad-split schemes only refine into quads"); - [ - uvs[fv[0] as usize], - uvs[fv[1] as usize], - uvs[fv[2] as usize], - uvs[fv[3] as usize], - ] - }) - .collect(); - SubdivFaces { - base_face, - corner_uvs, - } - }); - - let uvs = req.uvs.as_ref().map(|c| match chart_channel { - Some(ch) => { - // Face-varying: refine the values level by level, snap them to - // the limit, then read each refined face's entries. - let mut values = c.values.to_vec(); - for l in 1..=level { - let mut refined = vec![[0.0f32; 2]; refiner.level(l).num_fvar_values(ch)]; - primvar.interpolate_face_varying(l, ch, &values, &mut refined); - values = refined; - } - let mut limit_uvs = vec![[0.0f32; 2]; values.len()]; - primvar.limit_face_varying(ch, &values, &mut limit_uvs); - let mut fv_indices = Vec::with_capacity(out_indices.len()); - for f in 0..n_faces { - fv_indices.extend(last.face_fvar_values(f, ch).iter().map(|&v| v as i32)); - } - RefinedUvs { - values: limit_uvs, - indices: Some(fv_indices), - face_varying: true, - } - } - None => { - // Vertex: one value per point, refined and limited exactly like - // the positions. - let mut values: Vec<[f32; 2]> = (0..points.len()) - .map(|p| c.values[c.value_index(p)]) - .collect(); - for l in 1..=level { - let mut refined = vec![[0.0f32; 2]; refiner.level(l).num_vertices()]; - primvar.interpolate(l, &values, &mut refined); - values = refined; - } - let mut limit_uvs = vec![[0.0f32; 2]; values.len()]; - primvar.limit(&values, &mut limit_uvs); - RefinedUvs { - values: limit_uvs, - indices: None, - face_varying: false, - } - } - }); - - // Everything that needed the refiner is extracted; drop it — every - // level's topology, a ×4/3 of the last — before the result's own copies - // of the last level are made, so the two never coexist. This is the - // third of the transient the kernel design record costed. (`primvar` - // only borrows it; its last use is above.) - drop(refiner); - - let normals = smooth_normals(&limit, &out_counts, &out_indices); - let points: Vec = limit - .into_iter() - .map(|p| Vec3f::from([p[0], p[1], p[2]])) - .collect(); - - Ok(SubdividedMesh { - points, - counts: out_counts, - indices: out_indices, - normals, - faces, - uvs, - }) -} - -/// Smooth per-vertex normals for a polygon mesh: each face accumulates its -/// *unnormalized* area vector (the sum of its fan's cross products — twice -/// the face normal scaled by area, so larger faces weigh more) onto **every** -/// vertex of the face — per-fan-triangle accumulation would weigh a vertex -/// by where it happens to sit in the fan. Every sum is then normalized -/// (zero-length sums fall back to +Y rather than yield NaNs; the kernel -/// treats shading normals as directions only). -/// -/// A malformed face — fewer than three corners, running past `indices`, or -/// naming a point that does not exist — contributes nothing, the faces -/// `triangulate` skips. Refined meshes never have one; an unvalidated cage -/// handed to the displacement pass may. -pub(crate) fn smooth_normals(verts: &[[f32; 3]], counts: &[i32], indices: &[i32]) -> Vec<[f32; 3]> { - let at = |i: usize| Vec3A::from_array(verts[i]); - let mut sums = vec![Vec3A::ZERO; verts.len()]; - let mut off = 0usize; - for &fc in counts { - let fc = fc.max(0) as usize; - let Some(face) = indices.get(off..off + fc) else { - break; - }; - off += fc; - if fc < 3 || face.iter().any(|&i| i < 0 || i as usize >= verts.len()) { - continue; - } - let v0 = at(face[0] as usize); - let mut area = Vec3A::ZERO; - for k in 1..fc - 1 { - let (i1, i2) = (face[k] as usize, face[k + 1] as usize); - area += (at(i1) - v0).cross(at(i2) - v0); - } - for &i in face { - sums[i as usize] += area; - } - } - sums.iter() - .map(|n| { - if n.length_squared() > 1e-20 { - n.normalize().to_array() - } else { - [0.0, 1.0, 0.0] - } - }) - .collect() -} - -/// Smooth per-vertex normals for an *unrefined* subdivision cage — what a -/// subdivision surface shades with at refinement level 0, as Hydra's Storm -/// does at low complexity: the cage's silhouette, the surface's shading. -/// `None` for a cage [`subdivide`] would refuse too; it then renders faceted. -pub(crate) fn smooth_cage_normals( - points: &[Vec3f], - counts: &[i32], - indices: &[i32], -) -> Option> { - validate_cage(points.len(), counts, indices).ok()?; - let verts: Vec<[f32; 3]> = points.iter().map(|p| [p.x, p.y, p.z]).collect(); - Some(smooth_normals(&verts, counts, indices)) -} - -/// The refiner indexes with `usize` counts and `u32` indices, and it cannot -/// skip a malformed face the way `triangulate` does — so the cage is checked -/// whole, up front. -fn validate_cage( - n_verts: usize, - counts: &[i32], - indices: &[i32], -) -> Result<(Vec, Vec), SubdivError> { - let mut total = 0usize; - let mut counts_us = Vec::with_capacity(counts.len()); - for (face, &fc) in counts.iter().enumerate() { - if fc < 3 { - return Err(SubdivError::BadTopology(format!( - "face {face} has {fc} vertices (need at least 3)" - ))); - } - counts_us.push(fc as usize); - total += fc as usize; - } - if total != indices.len() { - return Err(SubdivError::BadTopology(format!( - "faceVertexCounts sums to {total} but faceVertexIndices has {} entries", - indices.len() - ))); - } - let mut indices_u32 = Vec::with_capacity(indices.len()); - for &i in indices { - if i < 0 || i as usize >= n_verts { - return Err(SubdivError::BadTopology(format!( - "face vertex index {i} out of range (mesh has {n_verts} points)" - ))); - } - indices_u32.push(i as u32); - } - Ok((counts_us, indices_u32)) -} - -/// Expands USD crease runs into the per-edge vertex pairs the refiner wants. -/// A run of `n` vertices contributes `n - 1` edges; `sharpnesses` carries -/// either one value per run or one per edge. Sharpness 10 is USD's "as sharp -/// as possible", which is exactly `sdc::SHARPNESS_INFINITE`; anything at or -/// above it is clamped there. -fn expand_crease_runs( - indices: &[i32], - lengths: &[i32], - sharpnesses: &[f32], -) -> Result<(Vec<[u32; 2]>, Vec), SubdivError> { - if lengths.is_empty() { - return Ok((Vec::new(), Vec::new())); - } - let mut n_edges = 0usize; - let mut n_verts = 0usize; - for (run, &len) in lengths.iter().enumerate() { - if len < 2 { - return Err(SubdivError::BadTopology(format!( - "crease run {run} has length {len} (need at least 2 vertices)" - ))); - } - n_edges += len as usize - 1; - n_verts += len as usize; - } - if n_verts != indices.len() { - return Err(SubdivError::BadTopology(format!( - "creaseLengths sums to {n_verts} but creaseIndices has {} entries", - indices.len() - ))); - } - let per_run = sharpnesses.len() == lengths.len(); - if !per_run && sharpnesses.len() != n_edges { - return Err(SubdivError::BadTopology(format!( - "creaseSharpnesses has {} entries (want {} per-run or {n_edges} per-edge)", - sharpnesses.len(), - lengths.len() - ))); - } - let clamp = |s: f32| { - if s >= sdc::SHARPNESS_INFINITE { - sdc::SHARPNESS_INFINITE - } else { - s.max(0.0) - } - }; - let mut pairs = Vec::with_capacity(n_edges); - let mut weights = Vec::with_capacity(n_edges); - let mut off = 0usize; - let mut edge = 0usize; - for (run, &len) in lengths.iter().enumerate() { - for k in 0..len as usize - 1 { - let (a, b) = (indices[off + k], indices[off + k + 1]); - if a < 0 || b < 0 { - return Err(SubdivError::BadTopology(format!( - "crease run {run} has a negative vertex index" - ))); - } - pairs.push([a as u32, b as u32]); - weights.push(clamp(if per_run { - sharpnesses[run] - } else { - sharpnesses[edge] - })); - edge += 1; - } - off += len as usize; - } - Ok((pairs, weights)) -} - -fn validate_corners( - indices: &[i32], - sharpnesses: &[f32], -) -> Result<(Vec, Vec), SubdivError> { - if indices.len() != sharpnesses.len() { - return Err(SubdivError::BadTopology(format!( - "cornerIndices has {} entries but cornerSharpnesses has {}", - indices.len(), - sharpnesses.len() - ))); - } - let mut out = Vec::with_capacity(indices.len()); - for &i in indices { - if i < 0 { - return Err(SubdivError::BadTopology( - "cornerIndices has a negative vertex index".into(), - )); - } - out.push(i as u32); - } - let weights = sharpnesses - .iter() - .map(|&s| { - if s >= sdc::SHARPNESS_INFINITE { - sdc::SHARPNESS_INFINITE - } else { - s.max(0.0) - } - }) - .collect(); - Ok((out, weights)) -} - -/// The synthetic face-varying channel carrying each cage face's Ptex -/// parameterization: one value per face-vertex (`0..sum(counts)` in authored -/// order, so no value is shared across faces), quads seeded with the four -/// Ptex corners. Non-quad faces get zeros — their descendants are marked -/// unmappable regardless, the channel just has to be well-formed. -fn ptex_fvar_channel(counts: &[usize]) -> (Vec<[f32; 2]>, Vec) { - const QUAD: [[f32; 2]; 4] = [[0.0, 0.0], [1.0, 0.0], [1.0, 1.0], [0.0, 1.0]]; - let total: usize = counts.iter().sum(); - let mut values = Vec::with_capacity(total); - for &fc in counts { - if fc == 4 { - values.extend_from_slice(&QUAD); - } else { - values.extend(std::iter::repeat_n([0.0f32; 2], fc)); - } - } - let indices = (0..total as u32).collect(); - (values, indices) -} - -// --------------------------------------------------------------------------- -// Per-face adaptive tessellation -// --------------------------------------------------------------------------- - -/// Feature-adaptive isolation depth of per-face tessellation's patch table. -/// -/// Shallow on purpose. Under Catmull-Clark an all-triangle cage is irregular -/// everywhere (every split triangle's centre is a valence-3 vertex), so -/// isolation at depth `d` refines every face `d` times: at 3 the Moana ocean's -/// 684 416-triangle cage cost a 19.9 GiB transient. Gregory patches cover what -/// isolation leaves irregular. Measured on the all-extraordinary cube (edges of -/// length 2): exact at sampling depths up to the isolation depth, and within -/// 0.0185 of the uniform limit (0.9% of an edge) at a rate of 4, next to an -/// extraordinary vertex only; regular faces are exact B-spline patches at any -/// depth. -const ADAPTIVE_ISOLATION: usize = 1; - -/// Per triangle of a per-face tessellation: the cage face Ptex addresses and -/// the triangle's corners in that face's unit square. `None` for a triangle of -/// an `n`-gon, which Ptex does not address here (as for uniform refinement). -pub(crate) struct TessellatedFaces { - pub base_face: Vec>, - pub corner_uvs: Vec<[[f32; 2]; 3]>, -} - -/// What [`tessellate_adaptive`] needs to size one segment of the cage: the two -/// points that span it — for a cage edge its two cage vertices (the edges are -/// rated before any patch exists, so there is no limit point yet), for a spoke -/// of a refined `n`-gon its two limit end points — and it answers the segment's -/// projected length over the target (`ℓ · σ / t`). The distance and the frustum -/// test use the box around exactly these points; `ScreenRate::segment_at` pads -/// it by its own diagonal before culling, so a limit curve straying off its -/// chord is not culled at the frustum's edge. -pub(crate) type SegmentSize<'a> = dyn Fn(&[[f32; 3]]) -> f32 + 'a; - -/// A limit-surface tessellation, in [`SubdividedMesh`]'s shapes (all -/// triangles), with Ptex corners as [`TessellatedFaces`]. -pub(crate) struct TessellatedMesh { - pub points: Vec, - pub indices: Vec, - pub normals: Vec<[f32; 3]>, - pub faces: Option, - /// A `vertex` chart evaluated at every vertex. - pub uvs: Option>, - /// A `faceVarying` chart evaluated per Ptex face, so each side of a seam - /// keeps its own values: `values`, and per triangle corner (parallel to - /// `indices`) an index into them. - pub face_varying_uvs: Option<(Vec<[f32; 2]>, Vec)>, - /// The smallest and largest edge rate used, for the debug line. - pub rate_range: (u32, u32), - /// Cage edges and spokes by rate, binned by `ceil(log2(rate))`: 1, 2, - /// 3–4, 5–8, … - pub rate_bins: Vec, - /// Ptex faces tessellated. - pub ptex_faces: usize, - /// The shape of the selected faces' triangles: [`TriangleQuality`] bins - /// for the interior grids' and for the stitched rings'. - pub quality: [[u64; QUALITY_BINS]; 2], -} - -/// Bins of [`triangle_quality`]: `[0.9, 1]`, `[0.5, 0.9)`, `[0.1, 0.5)`, -/// `[0.01, 0.1)`, `[0, 0.01)`. -pub(crate) const QUALITY_BINS: usize = 5; - -/// A triangle's shape, `4√3 · area / Σ edge²`: 1 for an equilateral -/// triangle, toward 0 for a sliver. -pub(crate) fn triangle_quality(a: Vec3A, b: Vec3A, c: Vec3A) -> f32 { - let area2 = (b - a).cross(c - a).length(); // twice the area - let sum = (b - a).length_squared() + (c - b).length_squared() + (a - c).length_squared(); - if sum <= 0.0 { - return 0.0; - } - (2.0 * 3f32.sqrt() * area2 / sum).clamp(0.0, 1.0) -} - -/// The [`QUALITY_BINS`] bin of a quality. -pub(crate) fn quality_bin(q: f32) -> usize { - match q { - q if q >= 0.9 => 0, - q if q >= 0.5 => 1, - q if q >= 0.1 => 2, - q if q >= 0.01 => 3, - _ => 4, - } -} - -/// A point on an unrefined face's boundary: its vertex, its Ptex coordinate and its -/// face-varying chart value. -type RingPoint = (u32, [f32; 2], [f32; 2]); - -/// Who owns a vertex of the tessellation, so the faces that meet there share it. -#[derive(Clone, Copy, PartialEq, Eq, Hash)] -enum VertexKey { - Cage(u32), - /// Point `i` of cage edge `edge`, counted from its lower vertex. - Edge(u32, u32), - /// The centre of `n`-gon `face`. - Centre(u32), - /// Point `i` of spoke `k` of `n`-gon `face`, counted from the edge midpoint. - Spoke(u32, u32, u32), -} - -/// Tessellates a Catmull-Clark or bilinear cage per Ptex face. -/// -/// Every cage edge is rated from its two cage vertices (`segment_size`, then -/// [`tessellate::edge_rate`] under `max_level`). A face with an edge rated above -/// 1 is *selected*: only selected faces get limit patches -/// (`refine_adaptive_selected`, `create_with_options_selected`), and each of -/// their Ptex quads is gridded and stitched to its edges -/// ([`tessellate::tessellate_quad`]) on the limit surface. Every other face -/// renders its cage, smooth-shaded, as level 0 does — what MoonRay does with a -/// face whose tessellation factor is 0 — so the cost grows with the refined -/// area, not the cage. A cage vertex a selected face touches takes that face's -/// limit point, and every corner and edge point is evaluated once and shared, -/// so a selected and an unselected face meet without a crack. -/// -/// `req.level` is ignored. A `vertex` chart is evaluated with the positions' -/// basis; a `faceVarying` one with the patch table's face-varying patches, -/// under its `faceVaryingLinearInterpolation` (linear across an unselected -/// face). -pub(crate) fn tessellate_adaptive( - points: &[Vec3f], - counts: &[i32], - indices: &[i32], - req: &SubdivRequest, - max_level: u32, - segment_size: &SegmentSize<'_>, -) -> Result { - use super::tessellate::{PointKey, edge_rate, tessellate_quad}; - use opensubdiv_rs::far::{AdaptiveOptions, PatchMap, PatchTableFactory, PatchTableOptions}; - use std::collections::HashMap; - - debug_assert!( - req.scheme != SubdivScheme::Loop, - "Loop is the caller's to route" - ); - let (counts_us, indices_u32) = validate_cage(points.len(), counts, indices)?; - let (crease_pairs, crease_weights) = expand_crease_runs( - req.crease_indices, - req.crease_lengths, - req.crease_sharpnesses, - )?; - let corners = validate_corners(req.corner_indices, req.corner_sharpnesses)?; - let scheme = match req.scheme { - SubdivScheme::Bilinear => sdc::SchemeType::Bilinear, - _ => sdc::SchemeType::Catmark, - }; - let chart = req.uvs.filter(|c| c.face_varying); - let vertex_chart = req.uvs.filter(|c| !c.face_varying); - - // Faces, and the cage edges. - let n_faces = counts_us.len(); - let mut starts = Vec::with_capacity(n_faces); - let mut at = 0usize; - for &n in &counts_us { - starts.push(at); - at += n; - } - let face_verts = |f: usize| &indices_u32[starts[f]..starts[f] + counts_us[f]]; - let mut edge_ids: HashMap<(u32, u32), u32> = HashMap::new(); - let mut edges: Vec<(u32, u32)> = Vec::new(); - let mut face_edges: Vec = Vec::with_capacity(indices_u32.len()); - for f in 0..n_faces { - let fv = face_verts(f); - for k in 0..fv.len() { - let (a, b) = (fv[k], fv[(k + 1) % fv.len()]); - let key = (a.min(b), a.max(b)); - let id = *edge_ids.entry(key).or_insert_with(|| { - edges.push(key); - (edges.len() - 1) as u32 - }); - face_edges.push(id); - } - } - let edges_of = |f: usize| &face_edges[starts[f]..starts[f] + counts_us[f]]; - - // Rates from the cage, then the selection: faces with a finer edge. - let base_rates: Vec = edges - .iter() - .map(|&(a, b)| { - let (pa, pb) = (base_point(points, a), base_point(points, b)); - edge_rate(segment_size(&[pa, pb]), max_level, false) - }) - .collect(); - let selected: Vec = (0..n_faces) - .map(|f| edges_of(f).iter().any(|&e| base_rates[e as usize] > 1)) - .collect(); - // A selected `n`-gon's Ptex quads split its edges at their midpoints, so - // those edges take an even rate — on both sides. - let mut edge_rates = base_rates.clone(); - for f in (0..n_faces).filter(|&f| selected[f] && counts_us[f] != 4) { - for &e in edges_of(f) { - let r = &mut edge_rates[e as usize]; - *r = (*r).max(2).next_multiple_of(2); - } - } - let rate_of = |a: u32, b: u32| { - let id = edge_ids[&(a.min(b), a.max(b))]; - (id, edge_rates[id as usize]) - }; - let canonical = |a: u32, b: u32, i: u32, n: u32| if a < b { i } else { n - i }; - - let mut rate_bins: Vec = Vec::new(); - let mut bin = |r: u32| { - let b = (32 - (r.max(1) - 1).leading_zeros()) as usize; - if rate_bins.len() <= b { - rate_bins.resize(b + 1, 0); - } - rate_bins[b] += 1; - }; - for &r in &edge_rates { - bin(r); - } - let selected_faces: Vec = (0..n_faces as u32) - .filter(|&f| selected[f as usize]) - .collect(); - let ptex_faces: usize = counts_us.iter().map(|&n| if n == 4 { 1 } else { n }).sum(); - - let mut vertex_of: HashMap = HashMap::new(); - let mut out_points: Vec = Vec::new(); - let mut out_normals: Vec<[f32; 3]> = Vec::new(); - let mut out_uvs: Vec<[f32; 2]> = Vec::new(); - let mut out_indices: Vec = Vec::new(); - let mut base_face: Vec> = Vec::new(); - let mut corner_uvs: Vec<[[f32; 2]; 3]> = Vec::new(); - let mut fv_values: Vec<[f32; 2]> = Vec::new(); - let mut fv_indices: Vec = Vec::new(); - // The face-varying limit value a refined face gave each boundary point, per - // chart side: a corner keyed by its cage vertex and chart value, an edge - // point by its key and the chart values at the edge's ends. An unrefined - // neighbour on the same side takes it, so the chart is continuous where the - // two meet. - let mut fvar_at: HashMap<(VertexKey, u32, u32), [f32; 2]> = HashMap::new(); - let chart_value = - |f: usize, k: usize| -> u32 { chart.map_or(0, |c| c.value_index(starts[f] + k) as u32) }; - let side = |a: u32, b: u32| (a.min(b), a.max(b)); - let (mut min_rate, mut max_rate) = (u32::MAX, 0u32); - let mut quality = [[0u64; QUALITY_BINS]; 2]; - - // --- Selected faces, on the limit surface ----------------------------- - if !selected_faces.is_empty() { - let options = sdc::Options::default() - .with_vtx_boundary_interpolation(req.boundary) - .with_fvar_linear_interpolation( - chart.map_or(sdc::FVarLinearInterpolation::All, |c| c.linear), - ); - let chart_indices: Vec = match chart { - Some(c) => (0..indices.len()) - .map(|fv| c.value_index(fv) as u32) - .collect(), - None => Vec::new(), - }; - let channels: Vec = chart - .map(|c| FVarChannelDescriptor::new(c.values.len(), &chart_indices)) - .into_iter() - .collect(); - let mut descriptor = TopologyDescriptor::new(points.len(), &counts_us, &indices_u32) - .with_creases(&crease_pairs, &crease_weights) - .with_corners(&corners.0, &corners.1); - if !channels.is_empty() { - descriptor = descriptor.with_fvar_channels(&channels); - } - let mut refiner = TopologyRefinerFactory::create(descriptor, scheme, options) - .map_err(SubdivError::Refine)?; - // A face regular in the vertex topology can be irregular in the - // chart's: isolate it too rather than capping it a level up. - let mut adaptive = - AdaptiveOptions::new(ADAPTIVE_ISOLATION).with_consider_fvar_channels(chart.is_some()); - adaptive.use_single_crease_patch = true; - refiner.refine_adaptive_selected(adaptive, &selected_faces); - // Smooth face-varying patches that follow the chart's own topology - // and rule, not OpenSubdiv's legacy linear ones. - let table_options = PatchTableOptions::new() - .with_fvar_tables(chart.is_some()) - .with_fvar_legacy_linear_patches(false); - let table = PatchTableFactory::create_with_options_selected( - &refiner, - &table_options, - &selected_faces, - ) - .map_err(SubdivError::Refine)?; - let map = PatchMap::new(&table); - let ptex_of = table.ptex_indices(); - - // Control values: every level's vertices, base first. - let primvar = PrimvarRefiner::new(&refiner); - let base: Vec<[f32; 3]> = points.iter().map(|p| [p.x, p.y, p.z]).collect(); - let mut control = base.clone(); - let mut level_vals = base; - for l in 1..=refiner.max_level() { - let mut refined = vec![[0.0f32; 3]; refiner.level(l).num_vertices()]; - primvar.interpolate(l, &level_vals, &mut refined); - control.extend_from_slice(&refined); - level_vals = refined; - } - let fvar_values: Option> = chart.map(|c| { - let mut values = c.values.to_vec(); - for level in primvar.interpolate_face_varying_all(0, c.values) { - values.extend_from_slice(&level); - } - values - }); - let uv_control: Option> = vertex_chart.map(|c| { - let mut control: Vec<[f32; 2]> = (0..points.len()) - .map(|v| c.values[c.value_index(v)]) - .collect(); - let mut level_vals = control.clone(); - for l in 1..=refiner.max_level() { - let mut refined = vec![[0.0f32; 2]; refiner.level(l).num_vertices()]; - primvar.interpolate(l, &level_vals, &mut refined); - control.extend_from_slice(&refined); - level_vals = refined; - } - control - }); - let eval = |ptex: usize, u: f32, v: f32| -> Option<([f32; 3], [f32; 3])> { - let patch = map.find_patch(ptex, u, v)?; - let (p, du, dv) = table.evaluate(patch, u, v, &control); - let n = Vec3A::from(du).cross(Vec3A::from(dv)); - let n = if n.length_squared() > 1e-24 { - n.normalize() - } else { - // A degenerate parameterization (a pole): the normal a hair inside. - let (u2, v2) = (u + (0.5 - u) * 1e-3, v + (0.5 - v) * 1e-3); - let patch = map.find_patch(ptex, u2, v2)?; - let (_, du, dv) = table.evaluate(patch, u2, v2, &control); - Vec3A::from(du).cross(Vec3A::from(dv)).normalize_or_zero() - }; - Some((p, n.to_array())) - }; - let eval_fvar = |ptex: usize, u: f32, v: f32| -> Option<[f32; 2]> { - let values = fvar_values.as_ref()?; - let patch = map.find_patch(ptex, u, v)?; - Some(table.evaluate_face_varying(patch, u, v, values, 0).0) - }; - let eval_uv = |ptex: usize, u: f32, v: f32| -> Option<[f32; 2]> { - let control = uv_control.as_ref()?; - let patch = map.find_patch(ptex, u, v)?; - Some(table.evaluate(patch, u, v, control).0) - }; - let mut emit = |key: Option, - ptex: usize, - uv: [f32; 2], - out_points: &mut Vec, - out_normals: &mut Vec<[f32; 3]>, - out_uvs: &mut Vec<[f32; 2]>| - -> Result { - if let Some(key) = key - && let Some(&v) = vertex_of.get(&key) - { - return Ok(v); - } - let (p, n) = eval(ptex, uv[0], uv[1]).ok_or_else(|| { - SubdivError::BadTopology(format!("no limit patch under Ptex face {ptex} at {uv:?}")) - })?; - let v = out_points.len() as u32; - out_points.push(Vec3f { - x: p[0], - y: p[1], - z: p[2], - }); - out_normals.push(n); - if uv_control.is_some() { - out_uvs.push(eval_uv(ptex, uv[0], uv[1]).unwrap_or([0.0, 0.0])); - } - if let Some(key) = key { - vertex_of.insert(key, v); - } - Ok(v) - }; - - for &f in &selected_faces { - let f = f as usize; - let fv = face_verts(f); - let first = ptex_of.face_id(f) as usize; - let n = fv.len(); - let quads: usize = if n == 4 { 1 } else { n }; - // Spoke rates of an `n`-gon: from the limit midpoint of edge k to - // the limit centre. - let spoke_rates: Vec = if n == 4 { - Vec::new() - } else { - let centre = eval(first, 1.0, 1.0).map(|e| e.0); - (0..n) - .map(|k| { - let mid = eval(first + k, 1.0, 0.0).map(|e| e.0); - match (mid, centre) { - (Some(m), Some(c)) => { - edge_rate(segment_size(&[m, c]), max_level, false) - } - _ => 1, - } - }) - .collect() - }; - for &r in &spoke_rates { - bin(r); - } - for k in 0..quads { - let ptex = first + k; - let mut rates = [0u32; 4]; - let mut edge_key: [Box VertexKey>; 4] = std::array::from_fn(|_| { - Box::new(|_| VertexKey::Cage(0)) as Box VertexKey> - }); - let corner_key: [VertexKey; 4] = if n == 4 { - for e in 0..4 { - let (a, b) = (fv[e], fv[(e + 1) % 4]); - let (id, r) = rate_of(a, b); - rates[e] = r; - edge_key[e] = Box::new(move |i| VertexKey::Edge(id, canonical(a, b, i, r))); - } - [0, 1, 2, 3].map(|c| VertexKey::Cage(fv[c])) - } else { - let (vk, vnext, vprev) = (fv[k], fv[(k + 1) % n], fv[(k + n - 1) % n]); - let (e_next, r_next) = rate_of(vk, vnext); - let (e_prev, r_prev) = rate_of(vprev, vk); - let (s_k, s_prev) = (spoke_rates[k], spoke_rates[(k + n - 1) % n]); - let (fi, ki, kp) = (f as u32, k as u32, ((k + n - 1) % n) as u32); - rates = [r_next / 2, s_k, s_prev, r_prev / 2]; - edge_key[0] = - Box::new(move |i| VertexKey::Edge(e_next, canonical(vk, vnext, i, r_next))); - edge_key[1] = Box::new(move |i| VertexKey::Spoke(fi, ki, i)); - edge_key[2] = Box::new(move |i| VertexKey::Spoke(fi, kp, s_prev - i)); - // From the midpoint toward vk: `half − i` segments from vk. - let half = r_prev / 2; - edge_key[3] = Box::new(move |i| { - let from_vk = half - i; - VertexKey::Edge(e_prev, canonical(vk, vprev, from_vk, r_prev)) - }); - [ - VertexKey::Cage(vk), - VertexKey::Edge(e_next, r_next / 2), - VertexKey::Centre(fi), - VertexKey::Edge(e_prev, r_prev / 2), - ] - }; - for &r in &rates { - min_rate = min_rate.min(r); - max_rate = max_rate.max(r); - } - let t = tessellate_quad(rates); - let mut local = Vec::with_capacity(t.points.len()); - for (uv, key) in t.points.iter().zip(&t.keys) { - let key = match *key { - PointKey::Corner(c) => Some(corner_key[c as usize]), - PointKey::Edge { edge, i } => Some(edge_key[edge as usize](i)), - PointKey::Interior => None, - }; - local.push(emit( - key, - ptex, - *uv, - &mut out_points, - &mut out_normals, - &mut out_uvs, - )?); - } - // The chart once per point of this Ptex face: a seam vertex - // shared with a face on the chart's other side takes this - // side's value. - let fv_first = fv_values.len() as i32; - if fvar_values.is_some() { - // The chart values at this Ptex quad's corners and along its - // cage-edge sides, for `fvar_at`. - type Side = Option<(u32, u32)>; - let (corner_side, edge_side): ([Side; 4], [Side; 4]) = if n == 4 { - let v = |c: usize| chart_value(f, c); - ( - [0, 1, 2, 3].map(|c| Some((v(c), v(c)))), - [0, 1, 2, 3].map(|e| Some(side(v(e), v((e + 1) % 4)))), - ) - } else { - let vk = chart_value(f, k); - let vn = chart_value(f, (k + 1) % n); - let vp = chart_value(f, (k + n - 1) % n); - ( - [Some((vk, vk)), Some(side(vk, vn)), None, Some(side(vp, vk))], - [Some(side(vk, vn)), None, None, Some(side(vp, vk))], - ) - }; - for (uv, key) in t.points.iter().zip(&t.keys) { - let value = eval_fvar(ptex, uv[0], uv[1]).unwrap_or([0.0, 0.0]); - fv_values.push(value); - let record = match *key { - PointKey::Corner(c) => { - corner_side[c as usize].map(|sd| (corner_key[c as usize], sd)) - } - PointKey::Edge { edge, i } => { - edge_side[edge as usize].map(|sd| (edge_key[edge as usize](i), sd)) - } - PointKey::Interior => None, - }; - if let Some((key, (a, b))) = record { - fvar_at.entry((key, a, b)).or_insert(value); - } - } - } - for (k, tri) in t.tris.iter().enumerate() { - let [a, b, c] = tri.map(|c| { - let p = out_points[local[c as usize] as usize]; - Vec3A::new(p.x, p.y, p.z) - }); - quality[usize::from(k >= t.stitched_from)] - [quality_bin(triangle_quality(a, b, c))] += 1; - for &c in tri { - out_indices.push(local[c as usize] as i32); - if fvar_values.is_some() { - fv_indices.push(fv_first + c as i32); - } - } - base_face.push((n == 4).then_some(f as u32)); - corner_uvs.push(tri.map(|c| t.points[c as usize])); - } - } - } - } - - // --- Unselected faces: the smooth cage -------------------------------- - if selected_faces.len() < n_faces { - let cage_normals = smooth_cage_normals(points, counts, indices) - .ok_or_else(|| SubdivError::BadTopology("cage normals".into()))?; - let corner_param = [[0.0f32, 0.0], [1.0, 0.0], [1.0, 1.0], [0.0, 1.0]]; - for f in (0..n_faces).filter(|&f| !selected[f]) { - let fv = face_verts(f); - let n = fv.len(); - // The face's boundary, corner by corner and along each edge's - // points, with its Ptex coordinate (quads) and chart value. - let mut ring: Vec = Vec::new(); - let chart_at = |k: usize| -> [f32; 2] { - chart.map_or([0.0, 0.0], |c| c.values[c.value_index(starts[f] + k)]) - }; - for k in 0..n { - let (a, b) = (fv[k], fv[(k + 1) % n]); - let v = match vertex_of.get(&VertexKey::Cage(a)) { - Some(&v) => v, - None => { - let v = out_points.len() as u32; - out_points.push(points[a as usize]); - out_normals.push(cage_normals[a as usize]); - if let Some(c) = vertex_chart { - out_uvs.push(c.values[c.value_index(a as usize)]); - } - vertex_of.insert(VertexKey::Cage(a), v); - v - } - }; - let (pa, pb) = if n == 4 { - (corner_param[k], corner_param[(k + 1) % 4]) - } else { - ([0.0, 0.0], [0.0, 0.0]) - }; - let (ia, ib) = (chart_value(f, k), chart_value(f, (k + 1) % n)); - let (ca, cb) = (chart_at(k), chart_at((k + 1) % n)); - let ca_limit = fvar_at - .get(&(VertexKey::Cage(a), ia, ia)) - .copied() - .unwrap_or(ca); - ring.push((v, pa, ca_limit)); - // An edge point here was placed by a selected neighbour (a - // midpoint its `n`-gon forced): use it, or the faces would - // meet at a T-junction. - let (id, r) = rate_of(a, b); - for i in 1..r { - let s = i as f32 / r as f32; - let key = VertexKey::Edge(id, canonical(a, b, i, r)); - let v = match vertex_of.get(&key) { - Some(&v) => v, - None => { - // Not reached: an edge above rate 1 has a selected - // face. Placed on the cage edge all the same. - let (p, q) = (points[a as usize], points[b as usize]); - let v = out_points.len() as u32; - out_points.push(Vec3f { - x: p.x + (q.x - p.x) * s, - y: p.y + (q.y - p.y) * s, - z: p.z + (q.z - p.z) * s, - }); - let (na, nb) = (cage_normals[a as usize], cage_normals[b as usize]); - out_normals.push( - Vec3A::from(na) - .lerp(Vec3A::from(nb), s) - .normalize_or_zero() - .to_array(), - ); - if let Some(c) = vertex_chart { - let (ua, ub) = ( - c.values[c.value_index(a as usize)], - c.values[c.value_index(b as usize)], - ); - out_uvs.push([ - ua[0] + (ub[0] - ua[0]) * s, - ua[1] + (ub[1] - ua[1]) * s, - ]); - } - vertex_of.insert(key, v); - v - } - }; - let lerp2 = |x: [f32; 2], y: [f32; 2]| { - [x[0] + (y[0] - x[0]) * s, x[1] + (y[1] - x[1]) * s] - }; - let (sa, sb) = side(ia, ib); - let value = fvar_at - .get(&(key, sa, sb)) - .copied() - .unwrap_or_else(|| lerp2(ca, cb)); - ring.push((v, lerp2(pa, pb), value)); - } - } - min_rate = min_rate.min(1); - max_rate = max_rate.max(1); - let mut push_tri = |tri: [&RingPoint; 3], - out_indices: &mut Vec, - fv_values: &mut Vec<[f32; 2]>, - fv_indices: &mut Vec| { - for c in tri { - out_indices.push(c.0 as i32); - if chart.is_some() { - fv_indices.push(fv_values.len() as i32); - fv_values.push(c.2); - } - } - base_face.push((n == 4).then_some(f as u32)); - corner_uvs.push(tri.map(|c| c.1)); - }; - if ring.len() == n { - // The cage polygon, fanned from its first corner as the - // importer triangulates. - for k in 1..n - 1 { - push_tri( - [&ring[0], &ring[k], &ring[k + 1]], - &mut out_indices, - &mut fv_values, - &mut fv_indices, - ); - } - } else { - // Edge points on the boundary: fan from the cage centroid. - let m = ring.len() as f32; - let centre_p = fv.iter().fold(Vec3A::ZERO, |acc, &v| { - let p = points[v as usize]; - acc + Vec3A::new(p.x, p.y, p.z) - }) / n as f32; - let centre_n = fv - .iter() - .fold(Vec3A::ZERO, |acc, &v| { - acc + Vec3A::from(cage_normals[v as usize]) - }) - .normalize_or_zero(); - let c = out_points.len() as u32; - out_points.push(Vec3f { - x: centre_p.x, - y: centre_p.y, - z: centre_p.z, - }); - out_normals.push(centre_n.to_array()); - if let Some(ch) = vertex_chart { - let mut uv = [0.0f32, 0.0]; - for &v in fv { - let x = ch.values[ch.value_index(v as usize)]; - uv[0] += x[0] / n as f32; - uv[1] += x[1] / n as f32; - } - out_uvs.push(uv); - } - let avg = |pick: fn(&RingPoint) -> [f32; 2]| { - let mut a = [0.0f32, 0.0]; - for r in &ring { - let x = pick(r); - a[0] += x[0] / m; - a[1] += x[1] / m; - } - a - }; - let centre = ( - c, - if n == 4 { [0.5, 0.5] } else { [0.0, 0.0] }, - avg(|r| r.2), - ); - for k in 0..ring.len() { - let next = &ring[(k + 1) % ring.len()]; - push_tri( - [¢re, &ring[k], next], - &mut out_indices, - &mut fv_values, - &mut fv_indices, - ); - } - } - } - } - - Ok(TessellatedMesh { - points: out_points, - indices: out_indices, - normals: out_normals, - faces: req.want_face_uvs.then_some(TessellatedFaces { - base_face, - corner_uvs, - }), - uvs: vertex_chart.is_some().then_some(out_uvs), - face_varying_uvs: chart.is_some().then_some((fv_values, fv_indices)), - rate_range: (min_rate.min(max_rate), max_rate), - rate_bins, - ptex_faces, - quality, - }) -} - -fn base_point(points: &[Vec3f], v: u32) -> [f32; 3] { - let p = points[v as usize]; - [p.x, p.y, p.z] -} - -#[cfg(test)] -mod tests { - use super::*; - - /// ±1 cube authored as six quads (the winding matches - /// `samples/subdivision.usda`). - fn cube() -> (Vec, Vec, Vec) { - let points = vec![ - Vec3f::from([-1.0, -1.0, 1.0]), - Vec3f::from([1.0, -1.0, 1.0]), - Vec3f::from([1.0, 1.0, 1.0]), - Vec3f::from([-1.0, 1.0, 1.0]), - Vec3f::from([-1.0, -1.0, -1.0]), - Vec3f::from([1.0, -1.0, -1.0]), - Vec3f::from([1.0, 1.0, -1.0]), - Vec3f::from([-1.0, 1.0, -1.0]), - ]; - let counts = vec![4; 6]; - let indices = vec![ - 0, 1, 2, 3, // +Z - 5, 4, 7, 6, // -Z - 4, 0, 3, 7, // -X - 1, 5, 6, 2, // +X - 3, 2, 6, 7, // +Y - 4, 5, 1, 0, // -Y - ]; - (points, counts, indices) - } - - fn request(level: u32) -> SubdivRequest<'static> { - SubdivRequest { - scheme: SubdivScheme::CatmullClark, - level, - boundary: sdc::VtxBoundaryInterpolation::EdgeAndCorner, - crease_indices: &[], - crease_lengths: &[], - crease_sharpnesses: &[], - corner_indices: &[], - corner_sharpnesses: &[], - want_face_uvs: false, - uvs: None, - } - } - - // --- Per-face adaptive tessellation ----------------------------------- - - /// Every undirected edge of a closed tessellation, used once each way. - fn assert_closed_and_consistent(indices: &[i32], what: &str) { - let mut directed = std::collections::HashMap::<(i32, i32), u32>::new(); - for t in indices.chunks(3) { - for (a, b) in [(t[0], t[1]), (t[1], t[2]), (t[2], t[0])] { - *directed.entry((a, b)).or_default() += 1; - } - } - for (&(a, b), &n) in &directed { - assert_eq!(n, 1, "{what}: edge {a}->{b} used {n} times"); - assert!( - directed.contains_key(&(b, a)), - "{what}: edge {a}->{b} has no twin: a crack or a winding flip" - ); - } - } - - /// A segment-size closure giving every segment `rate` segments, whatever - /// its length (`rate − 0.5` rounds up to `rate`). - fn constant(rate: u32) -> impl Fn(&[[f32; 3]]) -> f32 { - move |_| rate as f32 - 0.5 - } - - /// Rates that vary across the mesh, from each segment's first point: 1 to - /// 7 segments, so neighbouring faces disagree on their other edges. - fn mixed(p: &[[f32; 3]]) -> f32 { - let h = (p[0][0] * 3.1 + p[0][1] * 1.7 + p[0][2] * 2.3).abs(); - (h * 10.0) % 7.0 + 0.5 - } - - /// A pentagonal prism: two pentagons and five quads, closed. - fn prism() -> (Vec, Vec, Vec) { - let mut points = Vec::new(); - for z in [-1.0f32, 1.0] { - for k in 0..5 { - let a = k as f32 * std::f32::consts::TAU / 5.0; - points.push(Vec3f::from([a.cos(), a.sin(), z])); - } - } - let mut counts = vec![5, 5]; - let mut indices = vec![4, 3, 2, 1, 0, 5, 6, 7, 8, 9]; - for k in 0..5 { - let k1 = (k + 1) % 5; - counts.push(4); - indices.extend_from_slice(&[k, k1, 5 + k1, 5 + k]); - } - (points, counts, indices) - } - - #[test] - fn a_closed_cage_tessellates_closed_at_mixed_rates() { - for (name, (points, counts, indices)) in [("cube", cube()), ("prism", prism())] { - for max in [1u32, 2, 3] { - let t = tessellate_adaptive(&points, &counts, &indices, &request(0), max, &mixed) - .unwrap_or_else(|e| panic!("{name}: {e}")); - assert_closed_and_consistent(&t.indices, &format!("{name} at max {max}")); - if max == 3 { - assert!(t.rate_range.0 < t.rate_range.1, "{name}: rates should vary"); - } - } - let t = tessellate_adaptive(&points, &counts, &indices, &request(0), 3, &constant(1)) - .unwrap(); - assert_closed_and_consistent(&t.indices, &format!("{name} at rate 1")); - } - } - - /// A face whose every edge is split once is not refined at all: it renders - /// its cage, smooth-shaded, as level 0 does — and no patch is built for it. - #[test] - fn rate_one_faces_render_their_smooth_cage() { - let (points, counts, indices) = cube(); - let t = - tessellate_adaptive(&points, &counts, &indices, &request(0), 3, &constant(1)).unwrap(); - assert_eq!(t.points.len(), 8, "one vertex per cage corner"); - assert_eq!(t.indices.len(), 6 * 2 * 3); - // The cage's own positions and level 0's smooth normals, in whatever - // order the faces emitted them. - let smooth = smooth_cage_normals(&points, &counts, &indices).unwrap(); - for (p, n) in t.points.iter().zip(&t.normals) { - let k = points.iter().position(|q| q == p).expect("a cage position"); - assert_eq!(*n, smooth[k], "the smooth cage normal at {p:?}"); - } - } - - /// The limit points of uniform level `level`, as a list. - fn uniform_points( - points: &[Vec3f], - counts: &[i32], - indices: &[i32], - level: u32, - ) -> Vec<[f32; 3]> { - let m = subdivide(points, counts, indices, &request(level)).unwrap(); - m.points.iter().map(|p| [p.x, p.y, p.z]).collect() - } - - /// For each point, the distance to the nearest of `to`. - fn worst_distance(from: &[Vec3f], to: &[[f32; 3]]) -> f32 { - from.iter() - .map(|p| { - to.iter() - .map(|q| Vec3A::new(p.x - q[0], p.y - q[1], p.z - q[2]).length()) - .fold(f32::MAX, f32::min) - }) - .fold(0.0, f32::max) - } - - /// Where a refined face meets a face left at its cage, the surface stays - /// closed: the shared corners take the refined face's limit points, and an - /// edge point a refined `n`-gon forced is used by its unrefined neighbour. - #[test] - fn refined_and_cage_faces_meet_closed() { - // Only edges touching the +X side are rated above 1. - let near_x = |p: &[[f32; 3]]| { - if p[0][0] > 0.5 && p[1][0] > 0.5 { - 5.5 - } else { - 0.5 - } - }; - for (name, (points, counts, indices)) in [("cube", cube()), ("prism", prism())] { - let t = tessellate_adaptive(&points, &counts, &indices, &request(0), 3, &near_x) - .unwrap_or_else(|e| panic!("{name}: {e}")); - assert_closed_and_consistent(&t.indices, &format!("{name}, partly refined")); - assert!(t.rate_range.1 > 1, "{name}: some face is refined"); - let all = tessellate_adaptive(&points, &counts, &indices, &request(0), 3, &constant(6)) - .unwrap(); - assert!( - t.indices.len() < all.indices.len(), - "{name}: the faces left at their cage cost fewer triangles" - ); - } - } - - #[test] - fn a_uniform_rate_is_uniform_refinement_on_regular_faces() { - // A 6×6 grid of quads with a bump: interior faces are regular. - let g = 6; - let mut points = Vec::new(); - for j in 0..=g { - for i in 0..=g { - let (x, y) = (i as f32, j as f32); - points.push(Vec3f::from([ - x, - y, - ((x * 0.9).sin() * (y * 0.7).cos()) * 0.5, - ])); - } - } - let mut counts = Vec::new(); - let mut indices = Vec::new(); - for j in 0..g { - for i in 0..g { - let a = j * (g + 1) + i; - counts.push(4); - indices.extend_from_slice(&[a, a + 1, a + g + 2, a + g + 1]); - } - } - for level in [1u32, 2, 3] { - let t = tessellate_adaptive( - &points, - &counts, - &indices, - &request(0), - level, - &constant(1 << level), - ) - .unwrap(); - let uniform = uniform_points(&points, &counts, &indices, level); - assert_eq!(t.points.len(), uniform.len(), "level {level}: vertex count"); - let d = worst_distance(&t.points, &uniform); - assert!( - d < 1e-4, - "level {level}: a vertex is {d} from the uniform limit" - ); - } - } - - #[test] - fn near_extraordinary_vertices_the_patches_approximate_the_limit() { - // (The bound below and ADAPTIVE_ISOLATION's doc are the measurement.) - let (points, counts, indices) = cube(); - for level in [1u32, 2, 3] { - let t = tessellate_adaptive( - &points, - &counts, - &indices, - &request(0), - level, - &constant(1 << level), - ) - .unwrap(); - let uniform = uniform_points(&points, &counts, &indices, level); - assert_eq!(t.points.len(), uniform.len()); - let d = worst_distance(&t.points, &uniform); - // Measured and pinned: the cube is all extraordinary corners, the - // Gregory patches' worst case. Down to the isolation depth the - // samples are refined vertices, exact limit points; below it the - // Gregory patches approximate the limit (see ADAPTIVE_ISOLATION). - let bound = if level as usize <= ADAPTIVE_ISOLATION { - 1e-6 - } else { - 0.03 - }; - assert!(d < bound, "level {level}: {d}"); - } - } - - #[test] - fn ptex_corners_land_on_their_vertices() { - let (points, counts, indices) = cube(); - let req = SubdivRequest { - want_face_uvs: true, - ..request(0) - }; - let t = tessellate_adaptive(&points, &counts, &indices, &req, 3, &mixed).unwrap(); - let faces = t.faces.as_ref().expect("face table"); - assert_eq!(faces.base_face.len() * 3, t.indices.len()); - // Re-evaluate each corner through the patch table of its own face: it - // must be the vertex the triangle indexes. - let (counts_us, indices_u32) = validate_cage(points.len(), &counts, &indices).unwrap(); - let descriptor = TopologyDescriptor::new(points.len(), &counts_us, &indices_u32); - let mut refiner = TopologyRefinerFactory::create( - descriptor, - sdc::SchemeType::Catmark, - sdc::Options::default() - .with_vtx_boundary_interpolation(sdc::VtxBoundaryInterpolation::EdgeAndCorner), - ) - .unwrap(); - let mut adaptive = opensubdiv_rs::far::AdaptiveOptions::new(ADAPTIVE_ISOLATION); - adaptive.use_single_crease_patch = true; - refiner.refine_adaptive(adaptive); - let table = opensubdiv_rs::far::PatchTableFactory::create(&refiner).unwrap(); - let map = opensubdiv_rs::far::PatchMap::new(&table); - let primvar = PrimvarRefiner::new(&refiner); - let mut control: Vec<[f32; 3]> = points.iter().map(|p| [p.x, p.y, p.z]).collect(); - let mut vals = control.clone(); - for l in 1..=refiner.max_level() { - let mut r = vec![[0.0f32; 3]; refiner.level(l).num_vertices()]; - primvar.interpolate(l, &vals, &mut r); - control.extend_from_slice(&r); - vals = r; - } - let mut worst = 0.0f32; - for (tri, (face, corners)) in t - .indices - .chunks(3) - .zip(faces.base_face.iter().zip(&faces.corner_uvs)) - { - let face = face.expect("every cube face is a quad") as usize; - let ptex = table.ptex_indices().face_id(face) as usize; - for (&v, uv) in tri.iter().zip(corners) { - let patch = map.find_patch(ptex, uv[0], uv[1]).unwrap(); - let (p, _, _) = table.evaluate(patch, uv[0], uv[1], &control); - let q = t.points[v as usize]; - worst = worst.max(Vec3A::new(p[0] - q.x, p[1] - q.y, p[2] - q.z).length()); - } - } - assert!(worst < 1e-5, "a Ptex corner is {worst} off its vertex"); - } - - /// A face-varying chart giving every cube face its own unit square (every - /// edge a seam): each triangle corner's UV is its own Ptex coordinate, - /// whichever face shares the vertex — at mixed rates, where a shared seam - /// vertex would otherwise take the other side's value. - #[test] - fn a_seamed_face_varying_chart_keeps_each_side() { - let (points, counts, indices) = cube(); - let square = [[0.0f32, 0.0], [1.0, 0.0], [1.0, 1.0], [0.0, 1.0]]; - let values: Vec<[f32; 2]> = (0..6).flat_map(|_| square).collect(); - let chart_indices: Vec = (0..24).collect(); - for linear in [ - sdc::FVarLinearInterpolation::All, - sdc::FVarLinearInterpolation::Boundaries, - ] { - let req = SubdivRequest { - want_face_uvs: true, - uvs: Some(UvChannel { - values: &values, - indices: Some(&chart_indices), - face_varying: true, - linear, - }), - ..request(0) - }; - let t = tessellate_adaptive(&points, &counts, &indices, &req, 3, &mixed).unwrap(); - let (fv, corners) = t.face_varying_uvs.as_ref().expect("a face-varying chart"); - let ptex = t.faces.as_ref().unwrap(); - assert_eq!(corners.len(), t.indices.len()); - let mut worst = 0.0f32; - for (k, &c) in corners.iter().enumerate() { - let want = ptex.corner_uvs[k / 3][k % 3]; - let got = fv[c as usize]; - worst = worst.max((got[0] - want[0]).abs().max((got[1] - want[1]).abs())); - } - // The chart is affine on each face, so even a smooth rule - // reproduces it. - assert!( - worst < 1e-5, - "{linear:?}: a corner's UV is {worst} off its Ptex coordinate" - ); - } - } - - #[test] - fn triangle_quality_is_one_for_equilateral_and_falls_for_slivers() { - let (a, b) = (Vec3A::ZERO, Vec3A::X); - let apex = Vec3A::new(0.5, 3f32.sqrt() / 2.0, 0.0); - assert!((triangle_quality(a, b, apex) - 1.0).abs() < 1e-5); - assert!(triangle_quality(a, b, Vec3A::new(0.5, 0.01, 0.0)) < 0.05); - assert_eq!(triangle_quality(a, a, a), 0.0); - assert_eq!(quality_bin(1.0), 0); - assert_eq!(quality_bin(0.005), 4); - } - - /// A smooth, non-affine face-varying chart with no seam, on a curved grid - /// whose +X half is refined: every vertex carries one chart value, whichever - /// face — refined or left at its cage — uses it. Under `cornersPlus1` the - /// limit chart differs from the authored values at interior vertices, so an - /// unrefined face must take its refined neighbour's value where they meet. - #[test] - fn a_smooth_chart_is_continuous_where_refined_and_cage_faces_meet() { - let g = 6; - let mut points = Vec::new(); - let mut values = Vec::new(); - for j in 0..=g { - for i in 0..=g { - let (x, y) = (i as f32, j as f32); - points.push(Vec3f::from([ - x, - y, - ((x * 0.9).sin() * (y * 0.7).cos()) * 0.5, - ])); - values.push([0.1 * x * x, (0.4 * y).sin()]); - } - } - let (mut counts, mut indices) = (Vec::new(), Vec::new()); - for j in 0..g { - for i in 0..g { - let a = j * (g + 1) + i; - counts.push(4); - indices.extend_from_slice(&[a, a + 1, a + g + 2, a + g + 1]); - } - } - let near_x = |p: &[[f32; 3]]| { - if p[0][0] > 3.5 || p[1][0] > 3.5 { - 3.5 - } else { - 0.5 - } - }; - for linear in [ - sdc::FVarLinearInterpolation::CornersPlus1, - sdc::FVarLinearInterpolation::All, - ] { - let req = SubdivRequest { - uvs: Some(UvChannel { - values: &values, - indices: Some(&indices), - face_varying: true, - linear, - }), - ..request(0) - }; - let t = tessellate_adaptive(&points, &counts, &indices, &req, 3, &near_x).unwrap(); - let (fv, corners) = t.face_varying_uvs.as_ref().unwrap(); - let mut at: std::collections::HashMap = Default::default(); - let mut worst = 0.0f32; - for (&v, &c) in t.indices.iter().zip(corners) { - let uv = fv[c as usize]; - let first = *at.entry(v).or_insert(uv); - worst = worst.max((first[0] - uv[0]).abs().max((first[1] - uv[1]).abs())); - } - assert!( - worst < 1e-5, - "{linear:?}: a vertex's chart value jumps by {worst}" - ); - assert!(t.rate_range.1 > 1, "some faces are refined"); - } - } - - #[test] - fn normals_point_out_of_a_closed_surface() { - for (name, (points, counts, indices)) in [("cube", cube()), ("prism", prism())] { - let t = - tessellate_adaptive(&points, &counts, &indices, &request(0), 2, &mixed).unwrap(); - for (p, n) in t.points.iter().zip(&t.normals) { - let out = Vec3A::new(p.x, p.y, p.z).dot(Vec3A::from(*n)); - assert!(out > 0.0, "{name}: normal {n:?} at {p:?} points inward"); - } - } - } - - #[test] - fn cube_level_one_topology() { - let (points, counts, indices) = cube(); - let out = subdivide(&points, &counts, &indices, &request(1)).unwrap(); - assert_eq!(out.points.len(), 26, "8 corners + 12 edge + 6 face points"); - assert_eq!(out.counts.len(), 24, "each quad splits in four"); - assert!(out.counts.iter().all(|&c| c == 4)); - assert_eq!(out.indices.len(), 96); - assert_eq!(out.normals.len(), out.points.len()); - } - - #[test] - fn limit_shrinks_strictly_inside_the_cage() { - let (points, counts, indices) = cube(); - let out = subdivide(&points, &counts, &indices, &request(2)).unwrap(); - for p in &out.points { - for c in [p.x, p.y, p.z] { - assert!(c.abs() < 1.0, "limit point {p:?} not inside the cage"); - } - } - let max = out - .points - .iter() - .flat_map(|p| [p.x.abs(), p.y.abs(), p.z.abs()]) - .fold(0.0f32, f32::max); - assert!(max > 0.5, "limit surface collapsed too far ({max})"); - } - - #[test] - fn fully_creased_cube_keeps_its_cage() { - let (points, counts, indices) = cube(); - // All 12 edges as runs of 2 vertices, one sharpness per run. - let crease_indices: Vec = vec![ - 0, 1, 1, 2, 2, 3, 3, 0, // +Z ring - 4, 5, 5, 6, 6, 7, 7, 4, // -Z ring - 0, 4, 1, 5, 2, 6, 3, 7, // connecting edges - ]; - let crease_lengths = vec![2; 12]; - let crease_sharpnesses = vec![10.0f32; 12]; - let req = SubdivRequest { - crease_indices: &crease_indices, - crease_lengths: &crease_lengths, - crease_sharpnesses: &crease_sharpnesses, - ..request(2) - }; - let out = subdivide(&points, &counts, &indices, &req).unwrap(); - for axis in 0..3 { - let coords = out.points.iter().map(|p| [p.x, p.y, p.z][axis]); - let max = coords.clone().fold(f32::MIN, f32::max); - let min = coords.fold(f32::MAX, f32::min); - assert!((max - 1.0).abs() < 1e-5, "axis {axis} max {max}"); - assert!((min + 1.0).abs() < 1e-5, "axis {axis} min {min}"); - } - } - - #[test] - fn crease_runs_expand_per_run_and_per_edge() { - // One run of 3 vertices = 2 edges. - let (pairs, w) = expand_crease_runs(&[0, 1, 2], &[3], &[10.0]).unwrap(); - assert_eq!(pairs, vec![[0, 1], [1, 2]]); - assert_eq!(w, vec![10.0, 10.0], "per-run sharpness covers every edge"); - - let (_, w) = expand_crease_runs(&[0, 1, 2], &[3], &[2.0, 4.0]).unwrap(); - assert_eq!(w, vec![2.0, 4.0], "per-edge sharpness passes through"); - - assert!( - expand_crease_runs(&[0, 1, 2], &[3], &[1.0, 2.0, 3.0]).is_err(), - "3 sharpnesses fit neither 1 run nor 2 edges" - ); - assert!(expand_crease_runs(&[0, 1], &[3], &[1.0]).is_err()); - assert!(expand_crease_runs(&[0], &[1], &[1.0]).is_err()); - } - - #[test] - fn malformed_cages_are_rejected_whole() { - let (points, mut counts, indices) = cube(); - counts[0] = 2; - assert!(matches!( - subdivide(&points, &counts, &indices, &request(1)), - Err(SubdivError::BadTopology(_)) - )); - - let (points, counts, mut indices) = cube(); - indices[0] = 8; - assert!(subdivide(&points, &counts, &indices, &request(1)).is_err()); - - let (points, counts, _) = cube(); - assert!(subdivide(&points, &counts, &[0, 1, 2], &request(1)).is_err()); - } - - #[test] - fn level_one_face_uvs_tile_the_quadrants() { - let (points, counts, indices) = cube(); - let req = SubdivRequest { - want_face_uvs: true, - ..request(1) - }; - let out = subdivide(&points, &counts, &indices, &req).unwrap(); - let faces = out.faces.expect("face UVs were requested"); - assert_eq!(faces.base_face.len(), 24); - assert_eq!(faces.corner_uvs.len(), 24); - // Children of one parent are contiguous and in corner order, so the - // base_face map is 4 children per cage face... - for (child, &base) in faces.base_face.iter().enumerate() { - assert_eq!(base, Some((child / 4) as u32)); - } - // ...and each cage face's four children tile its unit square: every - // child covers a quarter, together they cover the whole, and every - // Ptex corner of the parent appears in exactly one child. - for parent in 0..6 { - let children = &faces.corner_uvs[parent * 4..parent * 4 + 4]; - let mut corner_hits = 0; - for quad in children { - let (mut umin, mut umax) = (f32::MAX, f32::MIN); - let (mut vmin, mut vmax) = (f32::MAX, f32::MIN); - for [u, v] in quad { - umin = umin.min(*u); - umax = umax.max(*u); - vmin = vmin.min(*v); - vmax = vmax.max(*v); - } - assert!((umax - umin - 0.5).abs() < 1e-6, "child spans half of u"); - assert!((vmax - vmin - 0.5).abs() < 1e-6, "child spans half of v"); - for corner in [[0.0, 0.0], [1.0, 0.0], [1.0, 1.0], [0.0, 1.0]] { - if quad - .iter() - .any(|c| (c[0] - corner[0]).abs() < 1e-6 && (c[1] - corner[1]).abs() < 1e-6) - { - corner_hits += 1; - } - } - } - assert_eq!(corner_hits, 4, "parent {parent}'s corners split 1:1"); - } - } - - #[test] - fn non_quad_base_faces_are_unmappable() { - // A quad with one corner cut off: one triangle + one pentagon. - let points = vec![ - Vec3f::from([0.0, 0.0, 0.0]), - Vec3f::from([2.0, 0.0, 0.0]), - Vec3f::from([2.0, 1.0, 0.0]), - Vec3f::from([1.0, 2.0, 0.0]), - Vec3f::from([0.0, 2.0, 0.0]), - Vec3f::from([2.0, 2.0, 0.0]), - ]; - let counts = vec![5, 3]; - let indices = vec![0, 1, 2, 3, 4, 2, 5, 3]; - let req = SubdivRequest { - want_face_uvs: true, - ..request(1) - }; - let out = subdivide(&points, &counts, &indices, &req).unwrap(); - let faces = out.faces.unwrap(); - assert_eq!(faces.base_face.len(), 8, "5 + 3 children"); - assert!(faces.base_face.iter().all(Option::is_none)); - } - - #[test] - fn smooth_cube_normals_point_along_the_corner_diagonals() { - let (points, counts, indices) = cube(); - let verts: Vec<[f32; 3]> = points.iter().map(|p| [p.x, p.y, p.z]).collect(); - let normals = smooth_normals(&verts, &counts, &indices); - for (v, n) in verts.iter().zip(&normals) { - let expect = Vec3A::from_array(*v).normalize(); - assert!( - Vec3A::from_array(*n).dot(expect) > 0.99, - "corner {v:?} normal {n:?} not along its diagonal" - ); - } - } - - #[test] - fn loop_refines_triangles() { - let points = vec![ - Vec3f::from([0.0, 0.0, 0.0]), - Vec3f::from([1.0, 0.0, 0.0]), - Vec3f::from([0.0, 1.0, 0.0]), - Vec3f::from([1.0, 1.0, 1.0]), - ]; - let counts = vec![3, 3]; - let indices = vec![0, 1, 2, 1, 3, 2]; - let req = SubdivRequest { - scheme: SubdivScheme::Loop, - ..request(1) - }; - let out = subdivide(&points, &counts, &indices, &req).unwrap(); - assert_eq!(out.counts.len(), 8, "each triangle splits in four"); - assert!(out.counts.iter().all(|&c| c == 3)); - } - - // ------------------------------------------------------------------- - // Memory probe - // ------------------------------------------------------------------- - - /// `System` wrapped in two counters, so the probe below measures - /// *requested* bytes — deterministic across platforms and allocators, - /// unlike RSS. Registered for the whole `crust_core` test binary (a - /// `#[global_allocator]` cannot be scoped tighter), which costs every - /// other test two relaxed atomics per allocation and changes nothing - /// else. - struct CountingAlloc; - - static LIVE: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); - static PEAK: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); - - // The crate is `deny(unsafe_code)`; this is the one exception, and it is - // scoped to the impl rather than to the module so anything else added - // nearby is still caught. `GlobalAlloc` cannot be implemented safely — - // that is the trait's contract, not a shortcut taken here — and the whole - // construct is `#[cfg(test)]`, so no `unsafe` reaches a shipped build. - #[allow(unsafe_code)] - unsafe impl std::alloc::GlobalAlloc for CountingAlloc { - unsafe fn alloc(&self, layout: std::alloc::Layout) -> *mut u8 { - use std::sync::atomic::Ordering::Relaxed; - let now = LIVE.fetch_add(layout.size(), Relaxed) + layout.size(); - PEAK.fetch_max(now, Relaxed); - unsafe { std::alloc::System.alloc(layout) } - } - unsafe fn dealloc(&self, ptr: *mut u8, layout: std::alloc::Layout) { - LIVE.fetch_sub(layout.size(), std::sync::atomic::Ordering::Relaxed); - unsafe { std::alloc::System.dealloc(ptr, layout) } - } - } - - #[global_allocator] - static COUNTING_ALLOC: CountingAlloc = CountingAlloc; - - /// An open N×N quad grid on the z = 0 plane — (N+1)² points, N² quads. - const ALL_FVAR_RULES: [sdc::FVarLinearInterpolation; 6] = [ - sdc::FVarLinearInterpolation::None, - sdc::FVarLinearInterpolation::CornersOnly, - sdc::FVarLinearInterpolation::CornersPlus1, - sdc::FVarLinearInterpolation::CornersPlus2, - sdc::FVarLinearInterpolation::Boundaries, - sdc::FVarLinearInterpolation::All, - ]; - - /// A flat 2×2 quad on z = 0 — one face, sharp corners. Catmull-Clark - /// reproduces affine data on it, positions and charts alike, so a chart - /// that is `point / 2` on the cage must still be `point / 2` at every - /// refined vertex. - fn flat_quad() -> (Vec, Vec, Vec) { - let points = vec![ - Vec3f::from([0.0, 0.0, 0.0]), - Vec3f::from([2.0, 0.0, 0.0]), - Vec3f::from([2.0, 2.0, 0.0]), - Vec3f::from([0.0, 2.0, 0.0]), - ]; - (points, vec![4], vec![0, 1, 2, 3]) - } - - const UNIT_SQUARE: [[f32; 2]; 4] = [[0.0, 0.0], [1.0, 0.0], [1.0, 1.0], [0.0, 1.0]]; - - /// The refined chart's value at refined face-vertex `fv`. - fn uv_at(uvs: &RefinedUvs, fv: usize, point: usize) -> [f32; 2] { - match &uvs.indices { - Some(idx) => uvs.values[idx[fv] as usize], - None => uvs.values[point], - } - } - - /// Asserts every refined face-vertex's UV is `f(its point)`. - fn assert_chart(out: &SubdividedMesh, f: impl Fn(Vec3f) -> [f32; 2], what: &str) { - let uvs = out.uvs.as_ref().expect("a chart was requested"); - for (fv, &p) in out.indices.iter().enumerate() { - let got = uv_at(uvs, fv, p as usize); - let want = f(out.points[p as usize]); - assert!( - (got[0] - want[0]).abs() < 1e-5 && (got[1] - want[1]).abs() < 1e-5, - "{what}: face-vertex {fv} at {:?} has uv {got:?}, expected {want:?}", - out.points[p as usize] - ); - } - } - - #[test] - fn face_varying_chart_refines_with_the_surface() { - let (points, counts, indices) = flat_quad(); - for linear in ALL_FVAR_RULES { - let req = SubdivRequest { - uvs: Some(UvChannel { - values: &UNIT_SQUARE, - indices: None, - face_varying: true, - linear, - }), - ..request(2) - }; - let out = subdivide(&points, &counts, &indices, &req).unwrap(); - let uvs = out.uvs.as_ref().unwrap(); - assert!(uvs.face_varying); - assert_eq!(uvs.indices.as_ref().unwrap().len(), out.indices.len()); - assert_chart(&out, |p| [p.x / 2.0, p.y / 2.0], &format!("{linear:?}")); - } - } - - /// Two quads sharing an edge, charted into two disjoint islands — the - /// shared edge is a UV seam. Each refined face must read its own island: - /// the seam vertices carry one value per side, never a blend of both. - #[test] - fn face_varying_seam_keeps_each_side_on_its_island() { - let points = vec![ - Vec3f::from([0.0, 0.0, 0.0]), - Vec3f::from([1.0, 0.0, 0.0]), - Vec3f::from([2.0, 0.0, 0.0]), - Vec3f::from([0.0, 1.0, 0.0]), - Vec3f::from([1.0, 1.0, 0.0]), - Vec3f::from([2.0, 1.0, 0.0]), - ]; - let counts = vec![4, 4]; - let indices = vec![0, 1, 4, 3, 1, 2, 5, 4]; - // Left island u = x, right island u = x + 4 (so [5, 6]). - let values = [ - [0.0, 0.0], - [1.0, 0.0], - [1.0, 1.0], - [0.0, 1.0], - [5.0, 0.0], - [6.0, 0.0], - [6.0, 1.0], - [5.0, 1.0], - ]; - let req = SubdivRequest { - uvs: Some(UvChannel { - values: &values, - indices: None, - face_varying: true, - linear: sdc::FVarLinearInterpolation::Boundaries, - }), - ..request(2) - }; - let out = subdivide(&points, &counts, &indices, &req).unwrap(); - let uvs = out.uvs.as_ref().unwrap(); - let mut offset = 0; - for &n in &out.counts { - let face = &out.indices[offset..offset + n as usize]; - let centre_x = face.iter().map(|&p| out.points[p as usize].x).sum::() / n as f32; - let shift = if centre_x < 1.0 { 0.0 } else { 4.0 }; - for (k, &p) in face.iter().enumerate() { - let got = uv_at(uvs, offset + k, p as usize); - let pt = out.points[p as usize]; - assert!( - (got[0] - (pt.x + shift)).abs() < 1e-5 && (got[1] - pt.y).abs() < 1e-5, - "face at x ~ {centre_x}: {pt:?} has uv {got:?}" - ); - } - offset += n as usize; - } - } - - /// `cornersPlus2` pins a concave UV corner (opensubdiv-rs ≥ 0.1.4). A 2×2 - /// grid whose faces F0-F2 form one L-shaped island and F3 another: at - /// the centre vertex the L's value spans three faces — a reflex corner — - /// and F3's spans one. `cornersPlus2` keeps the L's value where it was - /// authored, `cornersPlus1` smooths it along the island boundary. - #[test] - fn corners_plus2_pins_a_concave_uv_corner() { - let points: Vec = (0..3) - .flat_map(|j| (0..3).map(move |i| Vec3f::from([i as f32, j as f32, 0.0]))) - .collect(); - let counts = vec![4; 4]; - let indices = vec![0, 1, 4, 3, 1, 2, 5, 4, 3, 4, 7, 6, 4, 5, 8, 7]; - // The layout of opensubdiv-rs's own `L_ISLAND` fixture. - let values = [ - [0.0, 0.0], - [0.5, 0.0], - [1.0, 0.0], - [0.0, 0.5], - [0.5, 0.5], - [1.0, 0.4], - [0.0, 1.0], - [0.6, 1.0], - [0.5, 0.5], - [1.0, 0.5], - [1.0, 1.0], - [0.5, 1.0], - ]; - let value_indices = [0, 1, 4, 3, 1, 2, 5, 4, 3, 4, 7, 6, 8, 9, 10, 11]; - let centre_uv_on_the_l = |linear| { - let req = SubdivRequest { - uvs: Some(UvChannel { - values: &values, - indices: Some(&value_indices), - face_varying: true, - linear, - }), - ..request(1) - }; - let out = subdivide(&points, &counts, &indices, &req).unwrap(); - let uvs = out.uvs.as_ref().unwrap(); - // A refined face-vertex at the centre point, on a face of the L - // (every face but F3 = [1, 2]²). - let mut offset = 0; - for &n in &out.counts { - let face = &out.indices[offset..offset + n as usize]; - let centre = face.iter().fold([0.0f32; 2], |c, &p| { - let q = out.points[p as usize]; - [c[0] + q.x / n as f32, c[1] + q.y / n as f32] - }); - let on_l = !(centre[0] > 1.0 && centre[1] > 1.0); - for (k, &p) in face.iter().enumerate() { - let q = out.points[p as usize]; - if on_l && (q.x - 1.0).abs() < 1e-5 && (q.y - 1.0).abs() < 1e-5 { - return uv_at(uvs, offset + k, p as usize); - } - } - offset += n as usize; - } - panic!("no refined face-vertex at the centre on the L"); - }; - let pinned = centre_uv_on_the_l(sdc::FVarLinearInterpolation::CornersPlus2); - assert!( - (pinned[0] - 0.5).abs() < 1e-5 && (pinned[1] - 0.5).abs() < 1e-5, - "cornersPlus2 keeps the concave corner at (0.5, 0.5), got {pinned:?}" - ); - let smoothed = centre_uv_on_the_l(sdc::FVarLinearInterpolation::CornersPlus1); - assert!( - (smoothed[0] - 0.5).abs() > 1e-3 || (smoothed[1] - 0.5).abs() > 1e-3, - "cornersPlus1 smooths the concave corner, got {smoothed:?}" - ); - } - - #[test] - fn vertex_chart_refines_like_the_points() { - let (points, counts, indices) = flat_quad(); - // Authored through `:indices`, reversed, so the indirection is - // exercised: point p reads values[3 - p]. - let values = [ - UNIT_SQUARE[3], - UNIT_SQUARE[2], - UNIT_SQUARE[1], - UNIT_SQUARE[0], - ]; - let req = SubdivRequest { - uvs: Some(UvChannel { - values: &values, - indices: Some(&[3, 2, 1, 0]), - face_varying: false, - linear: sdc::FVarLinearInterpolation::CornersPlus1, - }), - ..request(2) - }; - let out = subdivide(&points, &counts, &indices, &req).unwrap(); - let uvs = out.uvs.as_ref().unwrap(); - assert!(!uvs.face_varying && uvs.indices.is_none()); - assert_eq!(uvs.values.len(), out.points.len()); - assert_chart(&out, |p| [p.x / 2.0, p.y / 2.0], "vertex"); - } - - #[test] - fn chart_indices_are_checked_against_the_values() { - let chart = |indices: Option<&'static [i32]>| UvChannel { - values: &UNIT_SQUARE, - indices, - face_varying: true, - linear: sdc::FVarLinearInterpolation::All, - }; - assert!(chart(None).is_well_formed(4)); - assert!(!chart(None).is_well_formed(5), "too few values"); - assert!(chart(Some(&[0, 1, 2, 3])).is_well_formed(4)); - assert!( - !chart(Some(&[0, 1, 2, 4])).is_well_formed(4), - "out of range" - ); - assert!(!chart(Some(&[0, 1, -1, 3])).is_well_formed(4), "negative"); - assert!( - !chart(Some(&[0, 1, 2])).is_well_formed(4), - "too few indices" - ); - } - - /// The face-varying rule is one per refiner, and an authored chart's rule - /// wins — so the Ptex face table of a UV-charted mesh must come out - /// bit-identical under every rule, or its lookups would drift off their - /// sub-face. Five rules get there by sharing the refiner (the channel is - /// invariant under them); `none`, which smooths face-varying corners, - /// gets there through its own chartless refinement. - #[test] - fn ptex_channel_is_invariant_under_every_fvar_rule() { - let (points, counts, indices) = cube(); - let reference = subdivide( - &points, - &counts, - &indices, - &SubdivRequest { - want_face_uvs: true, - ..request(2) - }, - ) - .unwrap() - .faces - .unwrap(); - let chart: Vec<[f32; 2]> = (0..6).flat_map(|_| UNIT_SQUARE).collect(); - for linear in ALL_FVAR_RULES { - let req = SubdivRequest { - want_face_uvs: true, - uvs: Some(UvChannel { - values: &chart, - indices: None, - face_varying: true, - linear, - }), - ..request(2) - }; - let faces = subdivide(&points, &counts, &indices, &req) - .unwrap() - .faces - .unwrap(); - assert_eq!(faces.base_face, reference.base_face, "{linear:?}"); - let bits = |f: &SubdivFaces| -> Vec { - f.corner_uvs - .iter() - .flatten() - .flatten() - .map(|c| c.to_bits()) - .collect() - }; - assert_eq!(bits(&faces), bits(&reference), "{linear:?}"); - } - } - - fn quad_grid(n: usize) -> (Vec, Vec, Vec) { - let mut points = Vec::with_capacity((n + 1) * (n + 1)); - for j in 0..=n { - for i in 0..=n { - points.push(Vec3f::from([i as f32, j as f32, 0.0])); - } - } - let mut counts = Vec::with_capacity(n * n); - let mut indices = Vec::with_capacity(4 * n * n); - for j in 0..n { - for i in 0..n { - let v0 = (j * (n + 1) + i) as i32; - let v1 = v0 + 1; - let v2 = v1 + (n + 1) as i32; - let v3 = v0 + (n + 1) as i32; - counts.push(4); - indices.extend_from_slice(&[v0, v1, v2, v3]); - } - } - (points, counts, indices) - } - - /// What subdivision costs in memory, measured at the `subdivide()` - /// boundary: `transient` is the peak of live requested bytes while it - /// runs (dominated by the refiner, which retains every level 0..L — - /// a ×4/3 geometric series over the last level — plus the position - /// copies at the tail of the function), `resident` is what the returned - /// [`SubdividedMesh`] itself holds. The ceilings pin the per-face costs - /// so a regression (say, an accidentally retained per-level buffer) - /// fails loudly; they sit ~25% above the values measured at the time of - /// writing, printed by the table for recalibration. - /// - /// Ignored because the counters are process-global: run it alone — - /// `cargo test -p crust-core --lib subdivision_memory_probe -- --ignored --nocapture --test-threads=1` - #[test] - #[ignore = "allocation probe; run alone with --ignored --nocapture --test-threads=1"] - fn subdivision_memory_probe() { - use std::sync::atomic::Ordering::Relaxed; - - const GRID: usize = 64; // 4096 cage quads - let (points, counts, indices) = quad_grid(GRID); - // A continuous chart, shared across faces the way an unwrapped UV set - // is: one value per point, addressed per face-vertex. - let chart_values: Vec<[f32; 2]> = points.iter().map(|p| [p.x, p.y]).collect(); - let chart = UvChannel { - values: &chart_values, - indices: Some(&indices), - face_varying: true, - linear: sdc::FVarLinearInterpolation::Boundaries, - }; - - println!( - "\n{:>5} {:>5} {:>14} {:>14} {:>10} {:>10}", - "level", "uvs", "faces", "transient", "B/face", "resid B/f" - ); - for level in 1..=4u32 { - for mode in ["no", "ptex", "chart"] { - let want_uvs = mode == "ptex"; - let req = SubdivRequest { - want_face_uvs: want_uvs, - uvs: (mode == "chart").then_some(chart), - ..request(level) - }; - - let before = LIVE.load(Relaxed); - PEAK.store(before, Relaxed); - let out = subdivide(&points, &counts, &indices, &req).unwrap(); - let peak = PEAK.load(Relaxed); - let after = LIVE.load(Relaxed); - - let n_faces = out.counts.len(); - let transient = peak - before; - let resident = after - before; - let per_face = transient as f64 / n_faces as f64; - let res_per_face = resident as f64 / n_faces as f64; - println!( - "{:>5} {:>5} {:>14} {:>14} {:>10.1} {:>10.1}", - level, mode, n_faces, transient, per_face, res_per_face - ); - - assert_eq!(n_faces, counts.len() * 4usize.pow(level)); - // Ceilings only bind once the per-cage-face constants have - // amortized away; shallow levels are all fixed overhead. - if level >= 3 { - // Measured on a 64×64 cage: ~313 B/face with no - // face-varying channel, ~540 with the Ptex one, ~517 with - // a shared UV chart; resident 44.0 / 84.0 / 68.1 (the - // Ptex table's 40 B/face: four corner UVs plus an - // `Option` base face; the normals are 12 B per - // vertex since `compact-triangle-storage`, 16 before). - // Ceilings ~20% above those, per mode, so no mode can - // hide under another's. - let (ceiling, resident_ceiling) = match mode { - "no" => (380.0, 53.0), - "ptex" => (650.0, 100.0), - _ => (620.0, 82.0), - }; - assert!( - per_face < ceiling, - "transient {per_face:.1} B/face at level {level} \ - (uvs: {mode}) exceeds the {ceiling} B/face ceiling" - ); - assert!( - res_per_face < resident_ceiling, - "resident {res_per_face:.1} B/face at level {level} \ - (uvs: {mode}) exceeds the {resident_ceiling} B/face ceiling" - ); - } - drop(out); - } - } - } -} diff --git a/crates/crust-core/src/scene/subdiv/adaptive.rs b/crates/crust-core/src/scene/subdiv/adaptive.rs new file mode 100644 index 00000000..3b7126e9 --- /dev/null +++ b/crates/crust-core/src/scene/subdiv/adaptive.rs @@ -0,0 +1,904 @@ +//! Per-face adaptive tessellation on the limit surface. + +use glam::Vec3A; +use opensubdiv_rs::far::{ + FVarChannelDescriptor, PrimvarRefiner, TopologyDescriptor, TopologyRefinerFactory, +}; +use opensubdiv_rs::sdc; +use openusd::gf::Vec3f; + +use std::collections::HashMap; + +use opensubdiv_rs::far::{AdaptiveOptions, PatchMap, PatchTableFactory, PatchTableOptions}; + +use super::super::tessellate::{PointKey, edge_rate, tessellate_quad}; +use super::UvChannel; +use super::normals::smooth_cage_normals; +use super::topology::{expand_crease_runs, validate_cage, validate_corners}; +use super::{SubdivError, SubdivRequest, SubdivScheme}; + +/// Feature-adaptive isolation depth of per-face tessellation's patch table. +/// +/// Shallow on purpose. Under Catmull-Clark an all-triangle cage is irregular +/// everywhere (every split triangle's centre is a valence-3 vertex), so +/// isolation at depth `d` refines every face `d` times: at 3 the Moana ocean's +/// 684 416-triangle cage cost a 19.9 GiB transient. Gregory patches cover what +/// isolation leaves irregular. Measured on the all-extraordinary cube (edges of +/// length 2): exact at sampling depths up to the isolation depth, and within +/// 0.0185 of the uniform limit (0.9% of an edge) at a rate of 4, next to an +/// extraordinary vertex only; regular faces are exact B-spline patches at any +/// depth. +pub(super) const ADAPTIVE_ISOLATION: usize = 1; + +/// Per triangle of a per-face tessellation: the cage face Ptex addresses and +/// the triangle's corners in that face's unit square. `None` for a triangle of +/// an `n`-gon, which Ptex does not address here (as for uniform refinement). +pub(crate) struct TessellatedFaces { + pub base_face: Vec>, + pub corner_uvs: Vec<[[f32; 2]; 3]>, +} + +/// What [`tessellate_adaptive`] needs to size one segment of the cage: the two +/// points that span it — for a cage edge its two cage vertices (the edges are +/// rated before any patch exists, so there is no limit point yet), for a spoke +/// of a refined `n`-gon its two limit end points — and it answers the segment's +/// projected length over the target (`ℓ · σ / t`). The distance and the frustum +/// test use the box around exactly these points; `ScreenRate::segment_at` pads +/// it by its own diagonal before culling, so a limit curve straying off its +/// chord is not culled at the frustum's edge. +pub(crate) type SegmentSize<'a> = dyn Fn(&[[f32; 3]]) -> f32 + 'a; + +/// A limit-surface tessellation, in [`SubdividedMesh`]'s shapes (all +/// triangles), with Ptex corners as [`TessellatedFaces`]. +pub(crate) struct TessellatedMesh { + pub points: Vec, + pub indices: Vec, + pub normals: Vec<[f32; 3]>, + pub faces: Option, + /// A `vertex` chart evaluated at every vertex. + pub uvs: Option>, + /// A `faceVarying` chart evaluated per Ptex face, so each side of a seam + /// keeps its own values: `values`, and per triangle corner (parallel to + /// `indices`) an index into them. + pub face_varying_uvs: Option<(Vec<[f32; 2]>, Vec)>, + /// The smallest and largest edge rate used, for the debug line. + pub rate_range: (u32, u32), + /// Cage edges and spokes by rate, binned by `ceil(log2(rate))`: 1, 2, + /// 3–4, 5–8, … + pub rate_bins: Vec, + /// Ptex faces tessellated. + pub ptex_faces: usize, + /// The shape of the selected faces' triangles: [`TriangleQuality`] bins + /// for the interior grids' and for the stitched rings'. + pub quality: [[u64; QUALITY_BINS]; 2], +} + +/// Bins of [`triangle_quality`]: `[0.9, 1]`, `[0.5, 0.9)`, `[0.1, 0.5)`, +/// `[0.01, 0.1)`, `[0, 0.01)`. +pub(crate) const QUALITY_BINS: usize = 5; + +/// A triangle's shape, `4√3 · area / Σ edge²`: 1 for an equilateral +/// triangle, toward 0 for a sliver. +pub(crate) fn triangle_quality(a: Vec3A, b: Vec3A, c: Vec3A) -> f32 { + let area2 = (b - a).cross(c - a).length(); // twice the area + let sum = (b - a).length_squared() + (c - b).length_squared() + (a - c).length_squared(); + if sum <= 0.0 { + return 0.0; + } + (2.0 * 3f32.sqrt() * area2 / sum).clamp(0.0, 1.0) +} + +/// The [`QUALITY_BINS`] bin of a quality. +pub(crate) fn quality_bin(q: f32) -> usize { + match q { + q if q >= 0.9 => 0, + q if q >= 0.5 => 1, + q if q >= 0.1 => 2, + q if q >= 0.01 => 3, + _ => 4, + } +} + +/// A point on an unrefined face's boundary: its vertex, its Ptex coordinate and its +/// face-varying chart value. +type RingPoint = (u32, [f32; 2], [f32; 2]); + +/// Who owns a vertex of the tessellation, so the faces that meet there share it. +#[derive(Clone, Copy, PartialEq, Eq, Hash)] +enum VertexKey { + Cage(u32), + /// Point `i` of cage edge `edge`, counted from its lower vertex. + Edge(u32, u32), + /// The centre of `n`-gon `face`. + Centre(u32), + /// Point `i` of spoke `k` of `n`-gon `face`, counted from the edge midpoint. + Spoke(u32, u32, u32), +} + +/// Tessellates a Catmull-Clark or bilinear cage per Ptex face. +/// +/// Every cage edge is rated from its two cage vertices (`segment_size`, then +/// [`tessellate::edge_rate`] under `max_level`). A face with an edge rated above +/// 1 is *selected*: only selected faces get limit patches +/// (`refine_adaptive_selected`, `create_with_options_selected`), and each of +/// their Ptex quads is gridded and stitched to its edges +/// ([`tessellate::tessellate_quad`]) on the limit surface. Every other face +/// renders its cage, smooth-shaded, as level 0 does — what MoonRay does with a +/// face whose tessellation factor is 0 — so the cost grows with the refined +/// area, not the cage. A cage vertex a selected face touches takes that face's +/// limit point, and every corner and edge point is evaluated once and shared, +/// so a selected and an unselected face meet without a crack. +/// +/// `req.level` is ignored. A `vertex` chart is evaluated with the positions' +/// basis; a `faceVarying` one with the patch table's face-varying patches, +/// under its `faceVaryingLinearInterpolation` (linear across an unselected +/// face). +pub(crate) fn tessellate_adaptive( + points: &[Vec3f], + counts: &[i32], + indices: &[i32], + req: &SubdivRequest, + max_level: u32, + segment_size: &SegmentSize<'_>, +) -> Result { + debug_assert!( + req.scheme != SubdivScheme::Loop, + "Loop is the caller's to route" + ); + let (counts_us, indices_u32) = validate_cage(points.len(), counts, indices)?; + let (crease_pairs, crease_weights) = expand_crease_runs( + req.crease_indices, + req.crease_lengths, + req.crease_sharpnesses, + )?; + let corners = validate_corners(req.corner_indices, req.corner_sharpnesses)?; + let sharpness = Sharpness { + crease_pairs: &crease_pairs, + crease_weights: &crease_weights, + corner_vertices: &corners.0, + corner_weights: &corners.1, + }; + let chart = req.uvs.filter(|c| c.face_varying); + let vertex_chart = req.uvs.filter(|c| !c.face_varying); + + let cage = Cage::new(points, counts_us, indices_u32); + let rating = EdgeRating::new(&cage, max_level, segment_size); + let mut out = Emitted::default(); + for &r in &rating.edge_rates { + out.bin(r); + } + let n_faces = cage.n_faces(); + let selected_faces: Vec = (0..n_faces as u32) + .filter(|&f| rating.selected[f as usize]) + .collect(); + let ptex_faces: usize = cage + .counts + .iter() + .map(|&n| if n == 4 { 1 } else { n }) + .sum(); + + let charts = Charts { + chart, + vertex_chart, + }; + if !selected_faces.is_empty() { + tessellate_selected( + &cage, + &rating, + req, + sharpness, + charts, + &selected_faces, + max_level, + segment_size, + &mut out, + )?; + } + if selected_faces.len() < n_faces { + tessellate_unselected(&cage, &rating, counts, indices, charts, &mut out)?; + } + + let Emitted { + points: out_points, + normals, + uvs, + indices: out_indices, + base_face, + corner_uvs, + fv_values, + fv_indices, + min_rate, + max_rate, + quality, + rate_bins, + .. + } = out; + Ok(TessellatedMesh { + points: out_points, + indices: out_indices, + normals, + faces: req.want_face_uvs.then_some(TessellatedFaces { + base_face, + corner_uvs, + }), + uvs: vertex_chart.is_some().then_some(uvs), + face_varying_uvs: chart.is_some().then_some((fv_values, fv_indices)), + rate_range: (min_rate.min(max_rate), max_rate), + rate_bins, + ptex_faces, + quality, + }) +} + +/// The cage's faces and its edges, each edge numbered once however many faces +/// share it. +struct Cage<'a> { + points: &'a [Vec3f], + counts: Vec, + indices: Vec, + /// Where each face's corners start in `indices`. + starts: Vec, + edge_ids: HashMap<(u32, u32), u32>, + /// Each edge as `(lower vertex, higher vertex)`. + edges: Vec<(u32, u32)>, + /// Per face-vertex, the edge from that corner to the next, parallel to + /// `indices`. + face_edges: Vec, +} + +impl<'a> Cage<'a> { + fn new(points: &'a [Vec3f], counts: Vec, indices: Vec) -> Self { + let mut starts = Vec::with_capacity(counts.len()); + let mut at = 0usize; + for &n in &counts { + starts.push(at); + at += n; + } + let mut edge_ids: HashMap<(u32, u32), u32> = HashMap::new(); + let mut edges: Vec<(u32, u32)> = Vec::new(); + let mut face_edges: Vec = Vec::with_capacity(indices.len()); + for (f, &n) in counts.iter().enumerate() { + let fv = &indices[starts[f]..starts[f] + n]; + for k in 0..fv.len() { + let (a, b) = (fv[k], fv[(k + 1) % fv.len()]); + let key = (a.min(b), a.max(b)); + let id = *edge_ids.entry(key).or_insert_with(|| { + edges.push(key); + (edges.len() - 1) as u32 + }); + face_edges.push(id); + } + } + Cage { + points, + counts, + indices, + starts, + edge_ids, + edges, + face_edges, + } + } + + fn n_faces(&self) -> usize { + self.counts.len() + } + + fn face_verts(&self, f: usize) -> &[u32] { + &self.indices[self.starts[f]..self.starts[f] + self.counts[f]] + } + + fn edges_of(&self, f: usize) -> &[u32] { + &self.face_edges[self.starts[f]..self.starts[f] + self.counts[f]] + } + + fn edge_id(&self, a: u32, b: u32) -> u32 { + self.edge_ids[&(a.min(b), a.max(b))] + } + + /// The chart value corner `k` of face `f` addresses (0 without a chart). + fn chart_value(&self, chart: Option>, f: usize, k: usize) -> u32 { + chart.map_or(0, |c| c.value_index(self.starts[f] + k) as u32) + } +} + +/// The edge rates, and which faces they select for limit patches. +struct EdgeRating { + /// Faces with an edge rated above 1. + selected: Vec, + /// Per cage edge, its rate — even on the edges of a selected `n`-gon. + edge_rates: Vec, +} + +impl EdgeRating { + /// Rates from the cage, then the selection: faces with a finer edge. + fn new(cage: &Cage<'_>, max_level: u32, segment_size: &SegmentSize<'_>) -> Self { + let base_rates: Vec = cage + .edges + .iter() + .map(|&(a, b)| { + let (pa, pb) = (base_point(cage.points, a), base_point(cage.points, b)); + edge_rate(segment_size(&[pa, pb]), max_level, false) + }) + .collect(); + let selected: Vec = (0..cage.n_faces()) + .map(|f| cage.edges_of(f).iter().any(|&e| base_rates[e as usize] > 1)) + .collect(); + // A selected `n`-gon's Ptex quads split its edges at their midpoints, + // so those edges take an even rate — on both sides. + let mut edge_rates = base_rates; + for f in (0..cage.n_faces()).filter(|&f| selected[f] && cage.counts[f] != 4) { + for &e in cage.edges_of(f) { + let r = &mut edge_rates[e as usize]; + *r = (*r).max(2).next_multiple_of(2); + } + } + EdgeRating { + selected, + edge_rates, + } + } + + /// The edge from `a` to `b`, and its rate. + fn rate_of(&self, cage: &Cage<'_>, a: u32, b: u32) -> (u32, u32) { + let id = cage.edge_id(a, b); + (id, self.edge_rates[id as usize]) + } +} + +/// Point `i` of an edge from `a` to `b` with `n` segments, counted from its +/// lower vertex. +fn canonical(a: u32, b: u32, i: u32, n: u32) -> u32 { + if a < b { i } else { n - i } +} + +/// An unordered pair, as a map key. +fn side(a: u32, b: u32) -> (u32, u32) { + (a.min(b), a.max(b)) +} + +/// The request's texture chart, by kind: a `faceVarying` one is evaluated per +/// Ptex face, a `vertex` one per vertex. +#[derive(Clone, Copy)] +struct Charts<'a> { + chart: Option>, + vertex_chart: Option>, +} + +/// The tessellation as it is emitted, both phases into the same arrays. +struct Emitted { + /// The vertex each shared corner, edge point, centre or spoke point became. + vertex_of: HashMap, + points: Vec, + normals: Vec<[f32; 3]>, + uvs: Vec<[f32; 2]>, + indices: Vec, + base_face: Vec>, + corner_uvs: Vec<[[f32; 2]; 3]>, + fv_values: Vec<[f32; 2]>, + fv_indices: Vec, + /// The face-varying limit value a refined face gave each boundary point, + /// per chart side: a corner keyed by its cage vertex and chart value, an + /// edge point by its key and the chart values at the edge's ends. An + /// unrefined neighbour on the same side takes it, so the chart is + /// continuous where the two meet. + fvar_at: HashMap<(VertexKey, u32, u32), [f32; 2]>, + min_rate: u32, + max_rate: u32, + quality: [[u64; QUALITY_BINS]; 2], + rate_bins: Vec, +} + +impl Default for Emitted { + fn default() -> Self { + Emitted { + vertex_of: HashMap::new(), + points: Vec::new(), + normals: Vec::new(), + uvs: Vec::new(), + indices: Vec::new(), + base_face: Vec::new(), + corner_uvs: Vec::new(), + fv_values: Vec::new(), + fv_indices: Vec::new(), + fvar_at: HashMap::new(), + min_rate: u32::MAX, + max_rate: 0, + quality: [[0; QUALITY_BINS]; 2], + rate_bins: Vec::new(), + } + } +} + +impl Emitted { + /// Counts one edge or spoke rate into [`TessellatedMesh::rate_bins`]. + fn bin(&mut self, r: u32) { + let b = (32 - (r.max(1) - 1).leading_zeros()) as usize; + if self.rate_bins.len() <= b { + self.rate_bins.resize(b + 1, 0); + } + self.rate_bins[b] += 1; + } + + fn see_rate(&mut self, r: u32) { + self.min_rate = self.min_rate.min(r); + self.max_rate = self.max_rate.max(r); + } +} + +/// The creases and corners the refiner is handed, validated up front. +#[derive(Clone, Copy)] +struct Sharpness<'a> { + crease_pairs: &'a [[u32; 2]], + crease_weights: &'a [f32], + corner_vertices: &'a [u32], + corner_weights: &'a [f32], +} + +/// The selected faces, on the limit surface: limit patches for exactly those +/// faces, and each of their Ptex quads gridded and stitched to its edges. +#[allow(clippy::too_many_arguments)] +fn tessellate_selected( + cage: &Cage<'_>, + rating: &EdgeRating, + req: &SubdivRequest, + sharpness: Sharpness<'_>, + Charts { + chart, + vertex_chart, + }: Charts<'_>, + selected_faces: &[u32], + max_level: u32, + segment_size: &SegmentSize<'_>, + out: &mut Emitted, +) -> Result<(), SubdivError> { + let points = cage.points; + let scheme = match req.scheme { + SubdivScheme::Bilinear => sdc::SchemeType::Bilinear, + _ => sdc::SchemeType::Catmark, + }; + let options = sdc::Options::default() + .with_vtx_boundary_interpolation(req.boundary) + .with_fvar_linear_interpolation( + chart.map_or(sdc::FVarLinearInterpolation::All, |c| c.linear), + ); + let chart_indices: Vec = match chart { + Some(c) => (0..cage.indices.len()) + .map(|fv| c.value_index(fv) as u32) + .collect(), + None => Vec::new(), + }; + let channels: Vec = chart + .map(|c| FVarChannelDescriptor::new(c.values.len(), &chart_indices)) + .into_iter() + .collect(); + let mut descriptor = TopologyDescriptor::new(points.len(), &cage.counts, &cage.indices) + .with_creases(sharpness.crease_pairs, sharpness.crease_weights) + .with_corners(sharpness.corner_vertices, sharpness.corner_weights); + if !channels.is_empty() { + descriptor = descriptor.with_fvar_channels(&channels); + } + let mut refiner = + TopologyRefinerFactory::create(descriptor, scheme, options).map_err(SubdivError::Refine)?; + // A face regular in the vertex topology can be irregular in the chart's: + // isolate it too rather than capping it a level up. + let mut adaptive = + AdaptiveOptions::new(ADAPTIVE_ISOLATION).with_consider_fvar_channels(chart.is_some()); + adaptive.use_single_crease_patch = true; + refiner.refine_adaptive_selected(adaptive, selected_faces); + // Smooth face-varying patches that follow the chart's own topology and + // rule, not OpenSubdiv's legacy linear ones. + let table_options = PatchTableOptions::new() + .with_fvar_tables(chart.is_some()) + .with_fvar_legacy_linear_patches(false); + let table = + PatchTableFactory::create_with_options_selected(&refiner, &table_options, selected_faces) + .map_err(SubdivError::Refine)?; + let map = PatchMap::new(&table); + let ptex_of = table.ptex_indices(); + + // Control values: every level's vertices, base first. + let primvar = PrimvarRefiner::new(&refiner); + let base: Vec<[f32; 3]> = points.iter().map(|p| [p.x, p.y, p.z]).collect(); + let mut control = base.clone(); + let mut level_vals = base; + for l in 1..=refiner.max_level() { + let mut refined = vec![[0.0f32; 3]; refiner.level(l).num_vertices()]; + primvar.interpolate(l, &level_vals, &mut refined); + control.extend_from_slice(&refined); + level_vals = refined; + } + let fvar_values: Option> = chart.map(|c| { + let mut values = c.values.to_vec(); + for level in primvar.interpolate_face_varying_all(0, c.values) { + values.extend_from_slice(&level); + } + values + }); + let uv_control: Option> = vertex_chart.map(|c| { + let mut control: Vec<[f32; 2]> = (0..points.len()) + .map(|v| c.values[c.value_index(v)]) + .collect(); + let mut level_vals = control.clone(); + for l in 1..=refiner.max_level() { + let mut refined = vec![[0.0f32; 2]; refiner.level(l).num_vertices()]; + primvar.interpolate(l, &level_vals, &mut refined); + control.extend_from_slice(&refined); + level_vals = refined; + } + control + }); + let eval = |ptex: usize, u: f32, v: f32| -> Option<([f32; 3], [f32; 3])> { + let patch = map.find_patch(ptex, u, v)?; + let (p, du, dv) = table.evaluate(patch, u, v, &control); + let n = Vec3A::from(du).cross(Vec3A::from(dv)); + let n = if n.length_squared() > 1e-24 { + n.normalize() + } else { + // A degenerate parameterization (a pole): the normal a hair inside. + let (u2, v2) = (u + (0.5 - u) * 1e-3, v + (0.5 - v) * 1e-3); + let patch = map.find_patch(ptex, u2, v2)?; + let (_, du, dv) = table.evaluate(patch, u2, v2, &control); + Vec3A::from(du).cross(Vec3A::from(dv)).normalize_or_zero() + }; + Some((p, n.to_array())) + }; + let eval_fvar = |ptex: usize, u: f32, v: f32| -> Option<[f32; 2]> { + let values = fvar_values.as_ref()?; + let patch = map.find_patch(ptex, u, v)?; + Some(table.evaluate_face_varying(patch, u, v, values, 0).0) + }; + let eval_uv = |ptex: usize, u: f32, v: f32| -> Option<[f32; 2]> { + let control = uv_control.as_ref()?; + let patch = map.find_patch(ptex, u, v)?; + Some(table.evaluate(patch, u, v, control).0) + }; + // The vertex for a point of Ptex face `ptex` at `uv`: the shared one when + // `key` names a point already emitted, else a new limit point. + let emit = |key: Option, + ptex: usize, + uv: [f32; 2], + out: &mut Emitted| + -> Result { + if let Some(key) = key + && let Some(&v) = out.vertex_of.get(&key) + { + return Ok(v); + } + let (p, n) = eval(ptex, uv[0], uv[1]).ok_or_else(|| { + SubdivError::BadTopology(format!("no limit patch under Ptex face {ptex} at {uv:?}")) + })?; + let v = out.points.len() as u32; + out.points.push(Vec3f { + x: p[0], + y: p[1], + z: p[2], + }); + out.normals.push(n); + if uv_control.is_some() { + out.uvs + .push(eval_uv(ptex, uv[0], uv[1]).unwrap_or([0.0, 0.0])); + } + if let Some(key) = key { + out.vertex_of.insert(key, v); + } + Ok(v) + }; + + for &f in selected_faces { + let f = f as usize; + let fv = cage.face_verts(f); + let first = ptex_of.face_id(f) as usize; + let n = fv.len(); + let quads: usize = if n == 4 { 1 } else { n }; + // Spoke rates of an `n`-gon: from the limit midpoint of edge k to the + // limit centre. + let spoke_rates: Vec = if n == 4 { + Vec::new() + } else { + let centre = eval(first, 1.0, 1.0).map(|e| e.0); + (0..n) + .map(|k| { + let mid = eval(first + k, 1.0, 0.0).map(|e| e.0); + match (mid, centre) { + (Some(m), Some(c)) => edge_rate(segment_size(&[m, c]), max_level, false), + _ => 1, + } + }) + .collect() + }; + for &r in &spoke_rates { + out.bin(r); + } + for k in 0..quads { + let ptex = first + k; + let mut rates = [0u32; 4]; + let mut edge_key: [Box VertexKey>; 4] = std::array::from_fn(|_| { + Box::new(|_| VertexKey::Cage(0)) as Box VertexKey> + }); + let corner_key: [VertexKey; 4] = if n == 4 { + for e in 0..4 { + let (a, b) = (fv[e], fv[(e + 1) % 4]); + let (id, r) = rating.rate_of(cage, a, b); + rates[e] = r; + edge_key[e] = Box::new(move |i| VertexKey::Edge(id, canonical(a, b, i, r))); + } + [0, 1, 2, 3].map(|c| VertexKey::Cage(fv[c])) + } else { + let (vk, vnext, vprev) = (fv[k], fv[(k + 1) % n], fv[(k + n - 1) % n]); + let (e_next, r_next) = rating.rate_of(cage, vk, vnext); + let (e_prev, r_prev) = rating.rate_of(cage, vprev, vk); + let (s_k, s_prev) = (spoke_rates[k], spoke_rates[(k + n - 1) % n]); + let (fi, ki, kp) = (f as u32, k as u32, ((k + n - 1) % n) as u32); + rates = [r_next / 2, s_k, s_prev, r_prev / 2]; + edge_key[0] = + Box::new(move |i| VertexKey::Edge(e_next, canonical(vk, vnext, i, r_next))); + edge_key[1] = Box::new(move |i| VertexKey::Spoke(fi, ki, i)); + edge_key[2] = Box::new(move |i| VertexKey::Spoke(fi, kp, s_prev - i)); + // From the midpoint toward vk: `half − i` segments from vk. + let half = r_prev / 2; + edge_key[3] = Box::new(move |i| { + let from_vk = half - i; + VertexKey::Edge(e_prev, canonical(vk, vprev, from_vk, r_prev)) + }); + [ + VertexKey::Cage(vk), + VertexKey::Edge(e_next, r_next / 2), + VertexKey::Centre(fi), + VertexKey::Edge(e_prev, r_prev / 2), + ] + }; + for &r in &rates { + out.see_rate(r); + } + let t = tessellate_quad(rates); + let mut local = Vec::with_capacity(t.points.len()); + for (uv, key) in t.points.iter().zip(&t.keys) { + let key = match *key { + PointKey::Corner(c) => Some(corner_key[c as usize]), + PointKey::Edge { edge, i } => Some(edge_key[edge as usize](i)), + PointKey::Interior => None, + }; + local.push(emit(key, ptex, *uv, out)?); + } + // The chart once per point of this Ptex face: a seam vertex shared + // with a face on the chart's other side takes this side's value. + let fv_first = out.fv_values.len() as i32; + if fvar_values.is_some() { + // The chart values at this Ptex quad's corners and along its + // cage-edge sides, for `fvar_at`. + type Side = Option<(u32, u32)>; + let (corner_side, edge_side): ([Side; 4], [Side; 4]) = if n == 4 { + let v = |c: usize| cage.chart_value(chart, f, c); + ( + [0, 1, 2, 3].map(|c| Some((v(c), v(c)))), + [0, 1, 2, 3].map(|e| Some(side(v(e), v((e + 1) % 4)))), + ) + } else { + let vk = cage.chart_value(chart, f, k); + let vn = cage.chart_value(chart, f, (k + 1) % n); + let vp = cage.chart_value(chart, f, (k + n - 1) % n); + ( + [Some((vk, vk)), Some(side(vk, vn)), None, Some(side(vp, vk))], + [Some(side(vk, vn)), None, None, Some(side(vp, vk))], + ) + }; + for (uv, key) in t.points.iter().zip(&t.keys) { + let value = eval_fvar(ptex, uv[0], uv[1]).unwrap_or([0.0, 0.0]); + out.fv_values.push(value); + let record = match *key { + PointKey::Corner(c) => { + corner_side[c as usize].map(|sd| (corner_key[c as usize], sd)) + } + PointKey::Edge { edge, i } => { + edge_side[edge as usize].map(|sd| (edge_key[edge as usize](i), sd)) + } + PointKey::Interior => None, + }; + if let Some((key, (a, b))) = record { + out.fvar_at.entry((key, a, b)).or_insert(value); + } + } + } + for (k, tri) in t.tris.iter().enumerate() { + let [a, b, c] = tri.map(|c| { + let p = out.points[local[c as usize] as usize]; + Vec3A::new(p.x, p.y, p.z) + }); + out.quality[usize::from(k >= t.stitched_from)] + [quality_bin(triangle_quality(a, b, c))] += 1; + for &c in tri { + out.indices.push(local[c as usize] as i32); + if fvar_values.is_some() { + out.fv_indices.push(fv_first + c as i32); + } + } + out.base_face.push((n == 4).then_some(f as u32)); + out.corner_uvs.push(tri.map(|c| t.points[c as usize])); + } + } + } + Ok(()) +} + +/// The unselected faces: the smooth cage, each polygon fanned as the importer +/// triangulates, through the edge points a selected neighbour placed so the +/// two meet without a T-junction. +fn tessellate_unselected( + cage: &Cage<'_>, + rating: &EdgeRating, + counts: &[i32], + indices: &[i32], + Charts { + chart, + vertex_chart, + }: Charts<'_>, + out: &mut Emitted, +) -> Result<(), SubdivError> { + let points = cage.points; + let cage_normals = smooth_cage_normals(points, counts, indices) + .ok_or_else(|| SubdivError::BadTopology("cage normals".into()))?; + let corner_param = [[0.0f32, 0.0], [1.0, 0.0], [1.0, 1.0], [0.0, 1.0]]; + for f in (0..cage.n_faces()).filter(|&f| !rating.selected[f]) { + let fv = cage.face_verts(f); + let n = fv.len(); + // The face's boundary, corner by corner and along each edge's points, + // with its Ptex coordinate (quads) and chart value. + let mut ring: Vec = Vec::new(); + let chart_at = |k: usize| -> [f32; 2] { + chart.map_or([0.0, 0.0], |c| c.values[c.value_index(cage.starts[f] + k)]) + }; + for k in 0..n { + let (a, b) = (fv[k], fv[(k + 1) % n]); + let v = match out.vertex_of.get(&VertexKey::Cage(a)) { + Some(&v) => v, + None => { + let v = out.points.len() as u32; + out.points.push(points[a as usize]); + out.normals.push(cage_normals[a as usize]); + if let Some(c) = vertex_chart { + out.uvs.push(c.values[c.value_index(a as usize)]); + } + out.vertex_of.insert(VertexKey::Cage(a), v); + v + } + }; + let (pa, pb) = if n == 4 { + (corner_param[k], corner_param[(k + 1) % 4]) + } else { + ([0.0, 0.0], [0.0, 0.0]) + }; + let (ia, ib) = ( + cage.chart_value(chart, f, k), + cage.chart_value(chart, f, (k + 1) % n), + ); + let (ca, cb) = (chart_at(k), chart_at((k + 1) % n)); + let ca_limit = out + .fvar_at + .get(&(VertexKey::Cage(a), ia, ia)) + .copied() + .unwrap_or(ca); + ring.push((v, pa, ca_limit)); + // An edge point here was placed by a selected neighbour (a midpoint + // its `n`-gon forced): use it, or the faces would meet at a + // T-junction. + let (id, r) = rating.rate_of(cage, a, b); + for i in 1..r { + let s = i as f32 / r as f32; + let key = VertexKey::Edge(id, canonical(a, b, i, r)); + let v = match out.vertex_of.get(&key) { + Some(&v) => v, + None => { + // Not reached: an edge above rate 1 has a selected + // face. Placed on the cage edge all the same. + let (p, q) = (points[a as usize], points[b as usize]); + let v = out.points.len() as u32; + out.points.push(Vec3f { + x: p.x + (q.x - p.x) * s, + y: p.y + (q.y - p.y) * s, + z: p.z + (q.z - p.z) * s, + }); + let (na, nb) = (cage_normals[a as usize], cage_normals[b as usize]); + out.normals.push( + Vec3A::from(na) + .lerp(Vec3A::from(nb), s) + .normalize_or_zero() + .to_array(), + ); + if let Some(c) = vertex_chart { + let (ua, ub) = ( + c.values[c.value_index(a as usize)], + c.values[c.value_index(b as usize)], + ); + out.uvs + .push([ua[0] + (ub[0] - ua[0]) * s, ua[1] + (ub[1] - ua[1]) * s]); + } + out.vertex_of.insert(key, v); + v + } + }; + let lerp2 = + |x: [f32; 2], y: [f32; 2]| [x[0] + (y[0] - x[0]) * s, x[1] + (y[1] - x[1]) * s]; + let (sa, sb) = side(ia, ib); + let value = out + .fvar_at + .get(&(key, sa, sb)) + .copied() + .unwrap_or_else(|| lerp2(ca, cb)); + ring.push((v, lerp2(pa, pb), value)); + } + } + out.see_rate(1); + let push_tri = |tri: [&RingPoint; 3], out: &mut Emitted| { + for c in tri { + out.indices.push(c.0 as i32); + if chart.is_some() { + out.fv_indices.push(out.fv_values.len() as i32); + out.fv_values.push(c.2); + } + } + out.base_face.push((n == 4).then_some(f as u32)); + out.corner_uvs.push(tri.map(|c| c.1)); + }; + if ring.len() == n { + // The cage polygon, fanned from its first corner as the importer + // triangulates. + for k in 1..n - 1 { + push_tri([&ring[0], &ring[k], &ring[k + 1]], out); + } + } else { + // Edge points on the boundary: fan from the cage centroid. + let m = ring.len() as f32; + let centre_p = fv.iter().fold(Vec3A::ZERO, |acc, &v| { + let p = points[v as usize]; + acc + Vec3A::new(p.x, p.y, p.z) + }) / n as f32; + let centre_n = fv + .iter() + .fold(Vec3A::ZERO, |acc, &v| { + acc + Vec3A::from(cage_normals[v as usize]) + }) + .normalize_or_zero(); + let c = out.points.len() as u32; + out.points.push(Vec3f { + x: centre_p.x, + y: centre_p.y, + z: centre_p.z, + }); + out.normals.push(centre_n.to_array()); + if let Some(ch) = vertex_chart { + let mut uv = [0.0f32, 0.0]; + for &v in fv { + let x = ch.values[ch.value_index(v as usize)]; + uv[0] += x[0] / n as f32; + uv[1] += x[1] / n as f32; + } + out.uvs.push(uv); + } + let avg = |pick: fn(&RingPoint) -> [f32; 2]| { + let mut a = [0.0f32, 0.0]; + for r in &ring { + let x = pick(r); + a[0] += x[0] / m; + a[1] += x[1] / m; + } + a + }; + let centre = ( + c, + if n == 4 { [0.5, 0.5] } else { [0.0, 0.0] }, + avg(|r| r.2), + ); + for k in 0..ring.len() { + let next = &ring[(k + 1) % ring.len()]; + push_tri([¢re, &ring[k], next], out); + } + } + } + Ok(()) +} + +fn base_point(points: &[Vec3f], v: u32) -> [f32; 3] { + let p = points[v as usize]; + [p.x, p.y, p.z] +} diff --git a/crates/crust-core/src/scene/subdiv/mod.rs b/crates/crust-core/src/scene/subdiv/mod.rs new file mode 100644 index 00000000..81efa6ff --- /dev/null +++ b/crates/crust-core/src/scene/subdiv/mod.rs @@ -0,0 +1,172 @@ +//! Subdivision-surface refinement for USD meshes, via the pure-Rust +//! [`opensubdiv-rs`] port of OpenSubdiv's Far/Sdc layers. +//! +//! The importer hands this module a base cage (points + faceVertexCounts + +//! faceVertexIndices, exactly as authored) and gets back a uniformly refined +//! mesh whose positions sit **on the limit surface** and whose vertices carry +//! smooth shading normals. Everything downstream — triangulation, interning, +//! instancing, baking — then treats the refined mesh like any other polygon +//! mesh. +//! +//! Ptex face ids index the *base cage*, so when the caller needs per-face +//! texturing the refinement also reports, per refined face, which cage face +//! it descends from and where its corners sit inside that face's unit square +//! ([`SubdivFaces`]). The sub-face UVs come from a synthetic face-varying +//! channel (each cage face owns four values at the Ptex corners, so every +//! edge of the channel is a face-varying boundary and it refines bilinearly +//! under every [`FVarLinearInterpolation`] rule but `None`, which is +//! refined apart) — the channel *is* the parameterization, so the smoothing +//! rules must never touch it. +//! +//! The authored texture chart ([`UvChannel`]) is refined with the surface: +//! a `faceVarying` chart as a real face-varying channel under the mesh's +//! `faceVaryingLinearInterpolation`, a `vertex` chart like the points. Both +//! are snapped to the limit, so a texel stays on the limit-surface point its +//! vertex was snapped to. +//! +//! [`opensubdiv-rs`]: https://github.com/doubleailes/OpenSubdiv-rs +//! [`FVarLinearInterpolation`]: sdc::FVarLinearInterpolation + +mod adaptive; +mod normals; +#[cfg(test)] +mod tests; +mod topology; +mod uniform; + +use opensubdiv_rs::sdc; +use openusd::gf::Vec3f; +use std::fmt; + +pub(crate) use adaptive::{QUALITY_BINS, TessellatedFaces, TessellatedMesh, tessellate_adaptive}; +pub(crate) use normals::{smooth_cage_normals, smooth_normals}; +pub(crate) use uniform::subdivide; + +/// The subdivision schemes the importer maps from `subdivisionScheme`. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub(crate) enum SubdivScheme { + CatmullClark, + Bilinear, + /// Loop subdivision refines triangles into triangles; the factory rejects + /// any non-triangular face, so the caller pre-checks the cage. + Loop, +} + +/// Everything `subdivide` needs beyond the cage arrays, borrowed straight +/// from the authored attributes. +#[derive(Clone, Copy)] +pub(crate) struct SubdivRequest<'a> { + pub scheme: SubdivScheme, + /// Uniform refinement depth, `>= 1` (level 0 never reaches this module). + pub level: u32, + pub boundary: sdc::VtxBoundaryInterpolation, + /// USD authors creases as runs of vertices: run `i` spans + /// `crease_lengths[i]` consecutive entries of `crease_indices` and + /// describes `crease_lengths[i] - 1` edges. + pub crease_indices: &'a [i32], + pub crease_lengths: &'a [i32], + /// One sharpness per run *or* one per edge — both are legal USD. + pub crease_sharpnesses: &'a [f32], + pub corner_indices: &'a [i32], + pub corner_sharpnesses: &'a [f32], + /// Build [`SubdivFaces`] (only wanted when the material samples a + /// per-face texture). + pub want_face_uvs: bool, + /// The authored texture chart to refine, when the material reads one. + pub uvs: Option>, +} + +/// An authored texture-coordinate primvar, as the importer read it. The +/// caller checks it with [`UvChannel::is_well_formed`] first — a refiner +/// cannot skip a bad value the way a triangle lookup can. +#[derive(Clone, Copy)] +pub(crate) struct UvChannel<'a> { + pub values: &'a [[f32; 2]], + /// The primvar's `:indices` into `values`, or `None` for direct + /// addressing. + pub indices: Option<&'a [i32]>, + /// `faceVarying` (one entry per face-vertex) rather than `vertex` (one + /// per point). + pub face_varying: bool, + /// The mesh's `faceVaryingLinearInterpolation`; only a `faceVarying` + /// chart reads it. + pub linear: sdc::FVarLinearInterpolation, +} + +impl UvChannel<'_> { + /// Whether every entry the cage addresses resolves to a value: + /// `n_entries` is the face-vertex count for a `faceVarying` chart, the + /// point count for a `vertex` one. + pub(crate) fn is_well_formed(&self, n_entries: usize) -> bool { + match self.indices { + Some(idx) => { + idx.len() >= n_entries + && idx[..n_entries] + .iter() + .all(|&i| i >= 0 && (i as usize) < self.values.len()) + } + None => self.values.len() >= n_entries, + } + } + + /// The value entry `i` (a face-vertex or a point) addresses. + pub(super) fn value_index(&self, i: usize) -> usize { + match self.indices { + Some(idx) => idx[i] as usize, + None => i, + } + } +} + +/// The refined chart, in the shape the importer's `UvSource` holds. +pub(crate) struct RefinedUvs { + pub values: Vec<[f32; 2]>, + /// Per refined face-vertex, into `values` — `Some` iff `face_varying`. + pub indices: Option>, + pub face_varying: bool, +} + +/// Per refined face: the base-cage face it descends from and its corner UVs +/// inside that face's unit square (Ptex convention: `v0=(0,0) v1=(1,0) +/// v2=(1,1) v3=(0,1)`). +pub(crate) struct SubdivFaces { + /// `None` for a face with no Ptex-addressable ancestor (its cage face + /// was not a quad — Ptex subfaces are out of scope, matching the + /// unsubdivided importer's treatment of n-gons). + pub base_face: Vec>, + /// Corner UVs of the refined quad, in `face_vertices` order. + pub corner_uvs: Vec<[[f32; 2]; 4]>, +} + +/// A refined mesh in the same array shapes `triangulate` consumes. +pub(crate) struct SubdividedMesh { + /// Limit-surface positions (uniform refinement, then limit snap). + pub points: Vec, + /// All 4s for Catmull-Clark/Bilinear, all 3s for Loop. + pub counts: Vec, + pub indices: Vec, + /// Smooth per-vertex shading normals, parallel to `points`, unpadded. + pub normals: Vec<[f32; 3]>, + /// `Some` iff the request asked for face UVs. + pub faces: Option, + /// `Some` iff the request carried a texture chart. + pub uvs: Option, +} + +#[derive(Debug)] +pub(crate) enum SubdivError { + /// The cage failed validation before it reached the refiner — unlike + /// `triangulate`, a topology refiner cannot skip a malformed face, so the + /// whole mesh degrades to its cage. + BadTopology(String), + Refine(opensubdiv_rs::far::Error), +} + +impl fmt::Display for SubdivError { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + SubdivError::BadTopology(why) => write!(f, "{why}"), + SubdivError::Refine(e) => write!(f, "{e}"), + } + } +} diff --git a/crates/crust-core/src/scene/subdiv/normals.rs b/crates/crust-core/src/scene/subdiv/normals.rs new file mode 100644 index 00000000..1295c1c6 --- /dev/null +++ b/crates/crust-core/src/scene/subdiv/normals.rs @@ -0,0 +1,66 @@ +//! Smooth per-vertex shading normals, for refined meshes and unrefined cages. + +use glam::Vec3A; +use openusd::gf::Vec3f; + +use super::topology::validate_cage; + +/// Smooth per-vertex normals for a polygon mesh: each face accumulates its +/// *unnormalized* area vector (the sum of its fan's cross products — twice +/// the face normal scaled by area, so larger faces weigh more) onto **every** +/// vertex of the face — per-fan-triangle accumulation would weigh a vertex +/// by where it happens to sit in the fan. Every sum is then normalized +/// (zero-length sums fall back to +Y rather than yield NaNs; the kernel +/// treats shading normals as directions only). +/// +/// A malformed face — fewer than three corners, running past `indices`, or +/// naming a point that does not exist — contributes nothing, the faces +/// `triangulate` skips. Refined meshes never have one; an unvalidated cage +/// handed to the displacement pass may. +pub(crate) fn smooth_normals(verts: &[[f32; 3]], counts: &[i32], indices: &[i32]) -> Vec<[f32; 3]> { + let at = |i: usize| Vec3A::from_array(verts[i]); + let mut sums = vec![Vec3A::ZERO; verts.len()]; + let mut off = 0usize; + for &fc in counts { + let fc = fc.max(0) as usize; + let Some(face) = indices.get(off..off + fc) else { + break; + }; + off += fc; + if fc < 3 || face.iter().any(|&i| i < 0 || i as usize >= verts.len()) { + continue; + } + let v0 = at(face[0] as usize); + let mut area = Vec3A::ZERO; + for k in 1..fc - 1 { + let (i1, i2) = (face[k] as usize, face[k + 1] as usize); + area += (at(i1) - v0).cross(at(i2) - v0); + } + for &i in face { + sums[i as usize] += area; + } + } + sums.iter() + .map(|n| { + if n.length_squared() > 1e-20 { + n.normalize().to_array() + } else { + [0.0, 1.0, 0.0] + } + }) + .collect() +} + +/// Smooth per-vertex normals for an *unrefined* subdivision cage — what a +/// subdivision surface shades with at refinement level 0, as Hydra's Storm +/// does at low complexity: the cage's silhouette, the surface's shading. +/// `None` for a cage [`subdivide`] would refuse too; it then renders faceted. +pub(crate) fn smooth_cage_normals( + points: &[Vec3f], + counts: &[i32], + indices: &[i32], +) -> Option> { + validate_cage(points.len(), counts, indices).ok()?; + let verts: Vec<[f32; 3]> = points.iter().map(|p| [p.x, p.y, p.z]).collect(); + Some(smooth_normals(&verts, counts, indices)) +} diff --git a/crates/crust-core/src/scene/subdiv/tests.rs b/crates/crust-core/src/scene/subdiv/tests.rs new file mode 100644 index 00000000..699d5aad --- /dev/null +++ b/crates/crust-core/src/scene/subdiv/tests.rs @@ -0,0 +1,1069 @@ +use glam::Vec3A; +use opensubdiv_rs::far::{PrimvarRefiner, TopologyDescriptor, TopologyRefinerFactory}; + +use super::adaptive::*; +use super::topology::*; +use super::*; + +/// ±1 cube authored as six quads (the winding matches +/// `samples/subdivision.usda`). +fn cube() -> (Vec, Vec, Vec) { + let points = vec![ + Vec3f::from([-1.0, -1.0, 1.0]), + Vec3f::from([1.0, -1.0, 1.0]), + Vec3f::from([1.0, 1.0, 1.0]), + Vec3f::from([-1.0, 1.0, 1.0]), + Vec3f::from([-1.0, -1.0, -1.0]), + Vec3f::from([1.0, -1.0, -1.0]), + Vec3f::from([1.0, 1.0, -1.0]), + Vec3f::from([-1.0, 1.0, -1.0]), + ]; + let counts = vec![4; 6]; + let indices = vec![ + 0, 1, 2, 3, // +Z + 5, 4, 7, 6, // -Z + 4, 0, 3, 7, // -X + 1, 5, 6, 2, // +X + 3, 2, 6, 7, // +Y + 4, 5, 1, 0, // -Y + ]; + (points, counts, indices) +} + +fn request(level: u32) -> SubdivRequest<'static> { + SubdivRequest { + scheme: SubdivScheme::CatmullClark, + level, + boundary: sdc::VtxBoundaryInterpolation::EdgeAndCorner, + crease_indices: &[], + crease_lengths: &[], + crease_sharpnesses: &[], + corner_indices: &[], + corner_sharpnesses: &[], + want_face_uvs: false, + uvs: None, + } +} + +// --- Per-face adaptive tessellation ----------------------------------- + +/// Every undirected edge of a closed tessellation, used once each way. +fn assert_closed_and_consistent(indices: &[i32], what: &str) { + let mut directed = std::collections::HashMap::<(i32, i32), u32>::new(); + for t in indices.chunks(3) { + for (a, b) in [(t[0], t[1]), (t[1], t[2]), (t[2], t[0])] { + *directed.entry((a, b)).or_default() += 1; + } + } + for (&(a, b), &n) in &directed { + assert_eq!(n, 1, "{what}: edge {a}->{b} used {n} times"); + assert!( + directed.contains_key(&(b, a)), + "{what}: edge {a}->{b} has no twin: a crack or a winding flip" + ); + } +} + +/// A segment-size closure giving every segment `rate` segments, whatever +/// its length (`rate − 0.5` rounds up to `rate`). +fn constant(rate: u32) -> impl Fn(&[[f32; 3]]) -> f32 { + move |_| rate as f32 - 0.5 +} + +/// Rates that vary across the mesh, from each segment's first point: 1 to +/// 7 segments, so neighbouring faces disagree on their other edges. +fn mixed(p: &[[f32; 3]]) -> f32 { + let h = (p[0][0] * 3.1 + p[0][1] * 1.7 + p[0][2] * 2.3).abs(); + (h * 10.0) % 7.0 + 0.5 +} + +/// A pentagonal prism: two pentagons and five quads, closed. +fn prism() -> (Vec, Vec, Vec) { + let mut points = Vec::new(); + for z in [-1.0f32, 1.0] { + for k in 0..5 { + let a = k as f32 * std::f32::consts::TAU / 5.0; + points.push(Vec3f::from([a.cos(), a.sin(), z])); + } + } + let mut counts = vec![5, 5]; + let mut indices = vec![4, 3, 2, 1, 0, 5, 6, 7, 8, 9]; + for k in 0..5 { + let k1 = (k + 1) % 5; + counts.push(4); + indices.extend_from_slice(&[k, k1, 5 + k1, 5 + k]); + } + (points, counts, indices) +} + +#[test] +fn a_closed_cage_tessellates_closed_at_mixed_rates() { + for (name, (points, counts, indices)) in [("cube", cube()), ("prism", prism())] { + for max in [1u32, 2, 3] { + let t = tessellate_adaptive(&points, &counts, &indices, &request(0), max, &mixed) + .unwrap_or_else(|e| panic!("{name}: {e}")); + assert_closed_and_consistent(&t.indices, &format!("{name} at max {max}")); + if max == 3 { + assert!(t.rate_range.0 < t.rate_range.1, "{name}: rates should vary"); + } + } + let t = + tessellate_adaptive(&points, &counts, &indices, &request(0), 3, &constant(1)).unwrap(); + assert_closed_and_consistent(&t.indices, &format!("{name} at rate 1")); + } +} + +/// A face whose every edge is split once is not refined at all: it renders +/// its cage, smooth-shaded, as level 0 does — and no patch is built for it. +#[test] +fn rate_one_faces_render_their_smooth_cage() { + let (points, counts, indices) = cube(); + let t = tessellate_adaptive(&points, &counts, &indices, &request(0), 3, &constant(1)).unwrap(); + assert_eq!(t.points.len(), 8, "one vertex per cage corner"); + assert_eq!(t.indices.len(), 6 * 2 * 3); + // The cage's own positions and level 0's smooth normals, in whatever + // order the faces emitted them. + let smooth = smooth_cage_normals(&points, &counts, &indices).unwrap(); + for (p, n) in t.points.iter().zip(&t.normals) { + let k = points.iter().position(|q| q == p).expect("a cage position"); + assert_eq!(*n, smooth[k], "the smooth cage normal at {p:?}"); + } +} + +/// The limit points of uniform level `level`, as a list. +fn uniform_points(points: &[Vec3f], counts: &[i32], indices: &[i32], level: u32) -> Vec<[f32; 3]> { + let m = subdivide(points, counts, indices, &request(level)).unwrap(); + m.points.iter().map(|p| [p.x, p.y, p.z]).collect() +} + +/// For each point, the distance to the nearest of `to`. +fn worst_distance(from: &[Vec3f], to: &[[f32; 3]]) -> f32 { + from.iter() + .map(|p| { + to.iter() + .map(|q| Vec3A::new(p.x - q[0], p.y - q[1], p.z - q[2]).length()) + .fold(f32::MAX, f32::min) + }) + .fold(0.0, f32::max) +} + +/// Where a refined face meets a face left at its cage, the surface stays +/// closed: the shared corners take the refined face's limit points, and an +/// edge point a refined `n`-gon forced is used by its unrefined neighbour. +#[test] +fn refined_and_cage_faces_meet_closed() { + // Only edges touching the +X side are rated above 1. + let near_x = |p: &[[f32; 3]]| { + if p[0][0] > 0.5 && p[1][0] > 0.5 { + 5.5 + } else { + 0.5 + } + }; + for (name, (points, counts, indices)) in [("cube", cube()), ("prism", prism())] { + let t = tessellate_adaptive(&points, &counts, &indices, &request(0), 3, &near_x) + .unwrap_or_else(|e| panic!("{name}: {e}")); + assert_closed_and_consistent(&t.indices, &format!("{name}, partly refined")); + assert!(t.rate_range.1 > 1, "{name}: some face is refined"); + let all = + tessellate_adaptive(&points, &counts, &indices, &request(0), 3, &constant(6)).unwrap(); + assert!( + t.indices.len() < all.indices.len(), + "{name}: the faces left at their cage cost fewer triangles" + ); + } +} + +#[test] +fn a_uniform_rate_is_uniform_refinement_on_regular_faces() { + // A 6×6 grid of quads with a bump: interior faces are regular. + let g = 6; + let mut points = Vec::new(); + for j in 0..=g { + for i in 0..=g { + let (x, y) = (i as f32, j as f32); + points.push(Vec3f::from([ + x, + y, + ((x * 0.9).sin() * (y * 0.7).cos()) * 0.5, + ])); + } + } + let mut counts = Vec::new(); + let mut indices = Vec::new(); + for j in 0..g { + for i in 0..g { + let a = j * (g + 1) + i; + counts.push(4); + indices.extend_from_slice(&[a, a + 1, a + g + 2, a + g + 1]); + } + } + for level in [1u32, 2, 3] { + let t = tessellate_adaptive( + &points, + &counts, + &indices, + &request(0), + level, + &constant(1 << level), + ) + .unwrap(); + let uniform = uniform_points(&points, &counts, &indices, level); + assert_eq!(t.points.len(), uniform.len(), "level {level}: vertex count"); + let d = worst_distance(&t.points, &uniform); + assert!( + d < 1e-4, + "level {level}: a vertex is {d} from the uniform limit" + ); + } +} + +#[test] +fn near_extraordinary_vertices_the_patches_approximate_the_limit() { + // (The bound below and ADAPTIVE_ISOLATION's doc are the measurement.) + let (points, counts, indices) = cube(); + for level in [1u32, 2, 3] { + let t = tessellate_adaptive( + &points, + &counts, + &indices, + &request(0), + level, + &constant(1 << level), + ) + .unwrap(); + let uniform = uniform_points(&points, &counts, &indices, level); + assert_eq!(t.points.len(), uniform.len()); + let d = worst_distance(&t.points, &uniform); + // Measured and pinned: the cube is all extraordinary corners, the + // Gregory patches' worst case. Down to the isolation depth the + // samples are refined vertices, exact limit points; below it the + // Gregory patches approximate the limit (see ADAPTIVE_ISOLATION). + let bound = if level as usize <= ADAPTIVE_ISOLATION { + 1e-6 + } else { + 0.03 + }; + assert!(d < bound, "level {level}: {d}"); + } +} + +#[test] +fn ptex_corners_land_on_their_vertices() { + let (points, counts, indices) = cube(); + let req = SubdivRequest { + want_face_uvs: true, + ..request(0) + }; + let t = tessellate_adaptive(&points, &counts, &indices, &req, 3, &mixed).unwrap(); + let faces = t.faces.as_ref().expect("face table"); + assert_eq!(faces.base_face.len() * 3, t.indices.len()); + // Re-evaluate each corner through the patch table of its own face: it + // must be the vertex the triangle indexes. + let (counts_us, indices_u32) = validate_cage(points.len(), &counts, &indices).unwrap(); + let descriptor = TopologyDescriptor::new(points.len(), &counts_us, &indices_u32); + let mut refiner = TopologyRefinerFactory::create( + descriptor, + sdc::SchemeType::Catmark, + sdc::Options::default() + .with_vtx_boundary_interpolation(sdc::VtxBoundaryInterpolation::EdgeAndCorner), + ) + .unwrap(); + let mut adaptive = opensubdiv_rs::far::AdaptiveOptions::new(ADAPTIVE_ISOLATION); + adaptive.use_single_crease_patch = true; + refiner.refine_adaptive(adaptive); + let table = opensubdiv_rs::far::PatchTableFactory::create(&refiner).unwrap(); + let map = opensubdiv_rs::far::PatchMap::new(&table); + let primvar = PrimvarRefiner::new(&refiner); + let mut control: Vec<[f32; 3]> = points.iter().map(|p| [p.x, p.y, p.z]).collect(); + let mut vals = control.clone(); + for l in 1..=refiner.max_level() { + let mut r = vec![[0.0f32; 3]; refiner.level(l).num_vertices()]; + primvar.interpolate(l, &vals, &mut r); + control.extend_from_slice(&r); + vals = r; + } + let mut worst = 0.0f32; + for (tri, (face, corners)) in t + .indices + .chunks(3) + .zip(faces.base_face.iter().zip(&faces.corner_uvs)) + { + let face = face.expect("every cube face is a quad") as usize; + let ptex = table.ptex_indices().face_id(face) as usize; + for (&v, uv) in tri.iter().zip(corners) { + let patch = map.find_patch(ptex, uv[0], uv[1]).unwrap(); + let (p, _, _) = table.evaluate(patch, uv[0], uv[1], &control); + let q = t.points[v as usize]; + worst = worst.max(Vec3A::new(p[0] - q.x, p[1] - q.y, p[2] - q.z).length()); + } + } + assert!(worst < 1e-5, "a Ptex corner is {worst} off its vertex"); +} + +/// A face-varying chart giving every cube face its own unit square (every +/// edge a seam): each triangle corner's UV is its own Ptex coordinate, +/// whichever face shares the vertex — at mixed rates, where a shared seam +/// vertex would otherwise take the other side's value. +#[test] +fn a_seamed_face_varying_chart_keeps_each_side() { + let (points, counts, indices) = cube(); + let square = [[0.0f32, 0.0], [1.0, 0.0], [1.0, 1.0], [0.0, 1.0]]; + let values: Vec<[f32; 2]> = (0..6).flat_map(|_| square).collect(); + let chart_indices: Vec = (0..24).collect(); + for linear in [ + sdc::FVarLinearInterpolation::All, + sdc::FVarLinearInterpolation::Boundaries, + ] { + let req = SubdivRequest { + want_face_uvs: true, + uvs: Some(UvChannel { + values: &values, + indices: Some(&chart_indices), + face_varying: true, + linear, + }), + ..request(0) + }; + let t = tessellate_adaptive(&points, &counts, &indices, &req, 3, &mixed).unwrap(); + let (fv, corners) = t.face_varying_uvs.as_ref().expect("a face-varying chart"); + let ptex = t.faces.as_ref().unwrap(); + assert_eq!(corners.len(), t.indices.len()); + let mut worst = 0.0f32; + for (k, &c) in corners.iter().enumerate() { + let want = ptex.corner_uvs[k / 3][k % 3]; + let got = fv[c as usize]; + worst = worst.max((got[0] - want[0]).abs().max((got[1] - want[1]).abs())); + } + // The chart is affine on each face, so even a smooth rule + // reproduces it. + assert!( + worst < 1e-5, + "{linear:?}: a corner's UV is {worst} off its Ptex coordinate" + ); + } +} + +#[test] +fn triangle_quality_is_one_for_equilateral_and_falls_for_slivers() { + let (a, b) = (Vec3A::ZERO, Vec3A::X); + let apex = Vec3A::new(0.5, 3f32.sqrt() / 2.0, 0.0); + assert!((triangle_quality(a, b, apex) - 1.0).abs() < 1e-5); + assert!(triangle_quality(a, b, Vec3A::new(0.5, 0.01, 0.0)) < 0.05); + assert_eq!(triangle_quality(a, a, a), 0.0); + assert_eq!(quality_bin(1.0), 0); + assert_eq!(quality_bin(0.005), 4); +} + +/// A smooth, non-affine face-varying chart with no seam, on a curved grid +/// whose +X half is refined: every vertex carries one chart value, whichever +/// face — refined or left at its cage — uses it. Under `cornersPlus1` the +/// limit chart differs from the authored values at interior vertices, so an +/// unrefined face must take its refined neighbour's value where they meet. +#[test] +fn a_smooth_chart_is_continuous_where_refined_and_cage_faces_meet() { + let g = 6; + let mut points = Vec::new(); + let mut values = Vec::new(); + for j in 0..=g { + for i in 0..=g { + let (x, y) = (i as f32, j as f32); + points.push(Vec3f::from([ + x, + y, + ((x * 0.9).sin() * (y * 0.7).cos()) * 0.5, + ])); + values.push([0.1 * x * x, (0.4 * y).sin()]); + } + } + let (mut counts, mut indices) = (Vec::new(), Vec::new()); + for j in 0..g { + for i in 0..g { + let a = j * (g + 1) + i; + counts.push(4); + indices.extend_from_slice(&[a, a + 1, a + g + 2, a + g + 1]); + } + } + let near_x = |p: &[[f32; 3]]| { + if p[0][0] > 3.5 || p[1][0] > 3.5 { + 3.5 + } else { + 0.5 + } + }; + for linear in [ + sdc::FVarLinearInterpolation::CornersPlus1, + sdc::FVarLinearInterpolation::All, + ] { + let req = SubdivRequest { + uvs: Some(UvChannel { + values: &values, + indices: Some(&indices), + face_varying: true, + linear, + }), + ..request(0) + }; + let t = tessellate_adaptive(&points, &counts, &indices, &req, 3, &near_x).unwrap(); + let (fv, corners) = t.face_varying_uvs.as_ref().unwrap(); + let mut at: std::collections::HashMap = Default::default(); + let mut worst = 0.0f32; + for (&v, &c) in t.indices.iter().zip(corners) { + let uv = fv[c as usize]; + let first = *at.entry(v).or_insert(uv); + worst = worst.max((first[0] - uv[0]).abs().max((first[1] - uv[1]).abs())); + } + assert!( + worst < 1e-5, + "{linear:?}: a vertex's chart value jumps by {worst}" + ); + assert!(t.rate_range.1 > 1, "some faces are refined"); + } +} + +#[test] +fn normals_point_out_of_a_closed_surface() { + for (name, (points, counts, indices)) in [("cube", cube()), ("prism", prism())] { + let t = tessellate_adaptive(&points, &counts, &indices, &request(0), 2, &mixed).unwrap(); + for (p, n) in t.points.iter().zip(&t.normals) { + let out = Vec3A::new(p.x, p.y, p.z).dot(Vec3A::from(*n)); + assert!(out > 0.0, "{name}: normal {n:?} at {p:?} points inward"); + } + } +} + +#[test] +fn cube_level_one_topology() { + let (points, counts, indices) = cube(); + let out = subdivide(&points, &counts, &indices, &request(1)).unwrap(); + assert_eq!(out.points.len(), 26, "8 corners + 12 edge + 6 face points"); + assert_eq!(out.counts.len(), 24, "each quad splits in four"); + assert!(out.counts.iter().all(|&c| c == 4)); + assert_eq!(out.indices.len(), 96); + assert_eq!(out.normals.len(), out.points.len()); +} + +#[test] +fn limit_shrinks_strictly_inside_the_cage() { + let (points, counts, indices) = cube(); + let out = subdivide(&points, &counts, &indices, &request(2)).unwrap(); + for p in &out.points { + for c in [p.x, p.y, p.z] { + assert!(c.abs() < 1.0, "limit point {p:?} not inside the cage"); + } + } + let max = out + .points + .iter() + .flat_map(|p| [p.x.abs(), p.y.abs(), p.z.abs()]) + .fold(0.0f32, f32::max); + assert!(max > 0.5, "limit surface collapsed too far ({max})"); +} + +#[test] +fn fully_creased_cube_keeps_its_cage() { + let (points, counts, indices) = cube(); + // All 12 edges as runs of 2 vertices, one sharpness per run. + let crease_indices: Vec = vec![ + 0, 1, 1, 2, 2, 3, 3, 0, // +Z ring + 4, 5, 5, 6, 6, 7, 7, 4, // -Z ring + 0, 4, 1, 5, 2, 6, 3, 7, // connecting edges + ]; + let crease_lengths = vec![2; 12]; + let crease_sharpnesses = vec![10.0f32; 12]; + let req = SubdivRequest { + crease_indices: &crease_indices, + crease_lengths: &crease_lengths, + crease_sharpnesses: &crease_sharpnesses, + ..request(2) + }; + let out = subdivide(&points, &counts, &indices, &req).unwrap(); + for axis in 0..3 { + let coords = out.points.iter().map(|p| [p.x, p.y, p.z][axis]); + let max = coords.clone().fold(f32::MIN, f32::max); + let min = coords.fold(f32::MAX, f32::min); + assert!((max - 1.0).abs() < 1e-5, "axis {axis} max {max}"); + assert!((min + 1.0).abs() < 1e-5, "axis {axis} min {min}"); + } +} + +#[test] +fn crease_runs_expand_per_run_and_per_edge() { + // One run of 3 vertices = 2 edges. + let (pairs, w) = expand_crease_runs(&[0, 1, 2], &[3], &[10.0]).unwrap(); + assert_eq!(pairs, vec![[0, 1], [1, 2]]); + assert_eq!(w, vec![10.0, 10.0], "per-run sharpness covers every edge"); + + let (_, w) = expand_crease_runs(&[0, 1, 2], &[3], &[2.0, 4.0]).unwrap(); + assert_eq!(w, vec![2.0, 4.0], "per-edge sharpness passes through"); + + assert!( + expand_crease_runs(&[0, 1, 2], &[3], &[1.0, 2.0, 3.0]).is_err(), + "3 sharpnesses fit neither 1 run nor 2 edges" + ); + assert!(expand_crease_runs(&[0, 1], &[3], &[1.0]).is_err()); + assert!(expand_crease_runs(&[0], &[1], &[1.0]).is_err()); +} + +#[test] +fn malformed_cages_are_rejected_whole() { + let (points, mut counts, indices) = cube(); + counts[0] = 2; + assert!(matches!( + subdivide(&points, &counts, &indices, &request(1)), + Err(SubdivError::BadTopology(_)) + )); + + let (points, counts, mut indices) = cube(); + indices[0] = 8; + assert!(subdivide(&points, &counts, &indices, &request(1)).is_err()); + + let (points, counts, _) = cube(); + assert!(subdivide(&points, &counts, &[0, 1, 2], &request(1)).is_err()); +} + +#[test] +fn level_one_face_uvs_tile_the_quadrants() { + let (points, counts, indices) = cube(); + let req = SubdivRequest { + want_face_uvs: true, + ..request(1) + }; + let out = subdivide(&points, &counts, &indices, &req).unwrap(); + let faces = out.faces.expect("face UVs were requested"); + assert_eq!(faces.base_face.len(), 24); + assert_eq!(faces.corner_uvs.len(), 24); + // Children of one parent are contiguous and in corner order, so the + // base_face map is 4 children per cage face... + for (child, &base) in faces.base_face.iter().enumerate() { + assert_eq!(base, Some((child / 4) as u32)); + } + // ...and each cage face's four children tile its unit square: every + // child covers a quarter, together they cover the whole, and every + // Ptex corner of the parent appears in exactly one child. + for parent in 0..6 { + let children = &faces.corner_uvs[parent * 4..parent * 4 + 4]; + let mut corner_hits = 0; + for quad in children { + let (mut umin, mut umax) = (f32::MAX, f32::MIN); + let (mut vmin, mut vmax) = (f32::MAX, f32::MIN); + for [u, v] in quad { + umin = umin.min(*u); + umax = umax.max(*u); + vmin = vmin.min(*v); + vmax = vmax.max(*v); + } + assert!((umax - umin - 0.5).abs() < 1e-6, "child spans half of u"); + assert!((vmax - vmin - 0.5).abs() < 1e-6, "child spans half of v"); + for corner in [[0.0, 0.0], [1.0, 0.0], [1.0, 1.0], [0.0, 1.0]] { + if quad + .iter() + .any(|c| (c[0] - corner[0]).abs() < 1e-6 && (c[1] - corner[1]).abs() < 1e-6) + { + corner_hits += 1; + } + } + } + assert_eq!(corner_hits, 4, "parent {parent}'s corners split 1:1"); + } +} + +#[test] +fn non_quad_base_faces_are_unmappable() { + // A quad with one corner cut off: one triangle + one pentagon. + let points = vec![ + Vec3f::from([0.0, 0.0, 0.0]), + Vec3f::from([2.0, 0.0, 0.0]), + Vec3f::from([2.0, 1.0, 0.0]), + Vec3f::from([1.0, 2.0, 0.0]), + Vec3f::from([0.0, 2.0, 0.0]), + Vec3f::from([2.0, 2.0, 0.0]), + ]; + let counts = vec![5, 3]; + let indices = vec![0, 1, 2, 3, 4, 2, 5, 3]; + let req = SubdivRequest { + want_face_uvs: true, + ..request(1) + }; + let out = subdivide(&points, &counts, &indices, &req).unwrap(); + let faces = out.faces.unwrap(); + assert_eq!(faces.base_face.len(), 8, "5 + 3 children"); + assert!(faces.base_face.iter().all(Option::is_none)); +} + +#[test] +fn smooth_cube_normals_point_along_the_corner_diagonals() { + let (points, counts, indices) = cube(); + let verts: Vec<[f32; 3]> = points.iter().map(|p| [p.x, p.y, p.z]).collect(); + let normals = smooth_normals(&verts, &counts, &indices); + for (v, n) in verts.iter().zip(&normals) { + let expect = Vec3A::from_array(*v).normalize(); + assert!( + Vec3A::from_array(*n).dot(expect) > 0.99, + "corner {v:?} normal {n:?} not along its diagonal" + ); + } +} + +#[test] +fn loop_refines_triangles() { + let points = vec![ + Vec3f::from([0.0, 0.0, 0.0]), + Vec3f::from([1.0, 0.0, 0.0]), + Vec3f::from([0.0, 1.0, 0.0]), + Vec3f::from([1.0, 1.0, 1.0]), + ]; + let counts = vec![3, 3]; + let indices = vec![0, 1, 2, 1, 3, 2]; + let req = SubdivRequest { + scheme: SubdivScheme::Loop, + ..request(1) + }; + let out = subdivide(&points, &counts, &indices, &req).unwrap(); + assert_eq!(out.counts.len(), 8, "each triangle splits in four"); + assert!(out.counts.iter().all(|&c| c == 3)); +} + +// ------------------------------------------------------------------- +// Memory probe +// ------------------------------------------------------------------- + +/// `System` wrapped in two counters, so the probe below measures +/// *requested* bytes — deterministic across platforms and allocators, +/// unlike RSS. Registered for the whole `crust_core` test binary (a +/// `#[global_allocator]` cannot be scoped tighter), which costs every +/// other test two relaxed atomics per allocation and changes nothing +/// else. +struct CountingAlloc; + +static LIVE: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); +static PEAK: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); + +// The crate is `deny(unsafe_code)`; this is the one exception, and it is +// scoped to the impl rather than to the module so anything else added +// nearby is still caught. `GlobalAlloc` cannot be implemented safely — +// that is the trait's contract, not a shortcut taken here — and the whole +// construct is `#[cfg(test)]`, so no `unsafe` reaches a shipped build. +#[allow(unsafe_code)] +unsafe impl std::alloc::GlobalAlloc for CountingAlloc { + unsafe fn alloc(&self, layout: std::alloc::Layout) -> *mut u8 { + use std::sync::atomic::Ordering::Relaxed; + let now = LIVE.fetch_add(layout.size(), Relaxed) + layout.size(); + PEAK.fetch_max(now, Relaxed); + unsafe { std::alloc::System.alloc(layout) } + } + unsafe fn dealloc(&self, ptr: *mut u8, layout: std::alloc::Layout) { + LIVE.fetch_sub(layout.size(), std::sync::atomic::Ordering::Relaxed); + unsafe { std::alloc::System.dealloc(ptr, layout) } + } +} + +#[global_allocator] +static COUNTING_ALLOC: CountingAlloc = CountingAlloc; + +/// An open N×N quad grid on the z = 0 plane — (N+1)² points, N² quads. +const ALL_FVAR_RULES: [sdc::FVarLinearInterpolation; 6] = [ + sdc::FVarLinearInterpolation::None, + sdc::FVarLinearInterpolation::CornersOnly, + sdc::FVarLinearInterpolation::CornersPlus1, + sdc::FVarLinearInterpolation::CornersPlus2, + sdc::FVarLinearInterpolation::Boundaries, + sdc::FVarLinearInterpolation::All, +]; + +/// A flat 2×2 quad on z = 0 — one face, sharp corners. Catmull-Clark +/// reproduces affine data on it, positions and charts alike, so a chart +/// that is `point / 2` on the cage must still be `point / 2` at every +/// refined vertex. +fn flat_quad() -> (Vec, Vec, Vec) { + let points = vec![ + Vec3f::from([0.0, 0.0, 0.0]), + Vec3f::from([2.0, 0.0, 0.0]), + Vec3f::from([2.0, 2.0, 0.0]), + Vec3f::from([0.0, 2.0, 0.0]), + ]; + (points, vec![4], vec![0, 1, 2, 3]) +} + +const UNIT_SQUARE: [[f32; 2]; 4] = [[0.0, 0.0], [1.0, 0.0], [1.0, 1.0], [0.0, 1.0]]; + +/// The refined chart's value at refined face-vertex `fv`. +fn uv_at(uvs: &RefinedUvs, fv: usize, point: usize) -> [f32; 2] { + match &uvs.indices { + Some(idx) => uvs.values[idx[fv] as usize], + None => uvs.values[point], + } +} + +/// Asserts every refined face-vertex's UV is `f(its point)`. +fn assert_chart(out: &SubdividedMesh, f: impl Fn(Vec3f) -> [f32; 2], what: &str) { + let uvs = out.uvs.as_ref().expect("a chart was requested"); + for (fv, &p) in out.indices.iter().enumerate() { + let got = uv_at(uvs, fv, p as usize); + let want = f(out.points[p as usize]); + assert!( + (got[0] - want[0]).abs() < 1e-5 && (got[1] - want[1]).abs() < 1e-5, + "{what}: face-vertex {fv} at {:?} has uv {got:?}, expected {want:?}", + out.points[p as usize] + ); + } +} + +#[test] +fn face_varying_chart_refines_with_the_surface() { + let (points, counts, indices) = flat_quad(); + for linear in ALL_FVAR_RULES { + let req = SubdivRequest { + uvs: Some(UvChannel { + values: &UNIT_SQUARE, + indices: None, + face_varying: true, + linear, + }), + ..request(2) + }; + let out = subdivide(&points, &counts, &indices, &req).unwrap(); + let uvs = out.uvs.as_ref().unwrap(); + assert!(uvs.face_varying); + assert_eq!(uvs.indices.as_ref().unwrap().len(), out.indices.len()); + assert_chart(&out, |p| [p.x / 2.0, p.y / 2.0], &format!("{linear:?}")); + } +} + +/// Two quads sharing an edge, charted into two disjoint islands — the +/// shared edge is a UV seam. Each refined face must read its own island: +/// the seam vertices carry one value per side, never a blend of both. +#[test] +fn face_varying_seam_keeps_each_side_on_its_island() { + let points = vec![ + Vec3f::from([0.0, 0.0, 0.0]), + Vec3f::from([1.0, 0.0, 0.0]), + Vec3f::from([2.0, 0.0, 0.0]), + Vec3f::from([0.0, 1.0, 0.0]), + Vec3f::from([1.0, 1.0, 0.0]), + Vec3f::from([2.0, 1.0, 0.0]), + ]; + let counts = vec![4, 4]; + let indices = vec![0, 1, 4, 3, 1, 2, 5, 4]; + // Left island u = x, right island u = x + 4 (so [5, 6]). + let values = [ + [0.0, 0.0], + [1.0, 0.0], + [1.0, 1.0], + [0.0, 1.0], + [5.0, 0.0], + [6.0, 0.0], + [6.0, 1.0], + [5.0, 1.0], + ]; + let req = SubdivRequest { + uvs: Some(UvChannel { + values: &values, + indices: None, + face_varying: true, + linear: sdc::FVarLinearInterpolation::Boundaries, + }), + ..request(2) + }; + let out = subdivide(&points, &counts, &indices, &req).unwrap(); + let uvs = out.uvs.as_ref().unwrap(); + let mut offset = 0; + for &n in &out.counts { + let face = &out.indices[offset..offset + n as usize]; + let centre_x = face.iter().map(|&p| out.points[p as usize].x).sum::() / n as f32; + let shift = if centre_x < 1.0 { 0.0 } else { 4.0 }; + for (k, &p) in face.iter().enumerate() { + let got = uv_at(uvs, offset + k, p as usize); + let pt = out.points[p as usize]; + assert!( + (got[0] - (pt.x + shift)).abs() < 1e-5 && (got[1] - pt.y).abs() < 1e-5, + "face at x ~ {centre_x}: {pt:?} has uv {got:?}" + ); + } + offset += n as usize; + } +} + +/// `cornersPlus2` pins a concave UV corner (opensubdiv-rs ≥ 0.1.4). A 2×2 +/// grid whose faces F0-F2 form one L-shaped island and F3 another: at +/// the centre vertex the L's value spans three faces — a reflex corner — +/// and F3's spans one. `cornersPlus2` keeps the L's value where it was +/// authored, `cornersPlus1` smooths it along the island boundary. +#[test] +fn corners_plus2_pins_a_concave_uv_corner() { + let points: Vec = (0..3) + .flat_map(|j| (0..3).map(move |i| Vec3f::from([i as f32, j as f32, 0.0]))) + .collect(); + let counts = vec![4; 4]; + let indices = vec![0, 1, 4, 3, 1, 2, 5, 4, 3, 4, 7, 6, 4, 5, 8, 7]; + // The layout of opensubdiv-rs's own `L_ISLAND` fixture. + let values = [ + [0.0, 0.0], + [0.5, 0.0], + [1.0, 0.0], + [0.0, 0.5], + [0.5, 0.5], + [1.0, 0.4], + [0.0, 1.0], + [0.6, 1.0], + [0.5, 0.5], + [1.0, 0.5], + [1.0, 1.0], + [0.5, 1.0], + ]; + let value_indices = [0, 1, 4, 3, 1, 2, 5, 4, 3, 4, 7, 6, 8, 9, 10, 11]; + let centre_uv_on_the_l = |linear| { + let req = SubdivRequest { + uvs: Some(UvChannel { + values: &values, + indices: Some(&value_indices), + face_varying: true, + linear, + }), + ..request(1) + }; + let out = subdivide(&points, &counts, &indices, &req).unwrap(); + let uvs = out.uvs.as_ref().unwrap(); + // A refined face-vertex at the centre point, on a face of the L + // (every face but F3 = [1, 2]²). + let mut offset = 0; + for &n in &out.counts { + let face = &out.indices[offset..offset + n as usize]; + let centre = face.iter().fold([0.0f32; 2], |c, &p| { + let q = out.points[p as usize]; + [c[0] + q.x / n as f32, c[1] + q.y / n as f32] + }); + let on_l = !(centre[0] > 1.0 && centre[1] > 1.0); + for (k, &p) in face.iter().enumerate() { + let q = out.points[p as usize]; + if on_l && (q.x - 1.0).abs() < 1e-5 && (q.y - 1.0).abs() < 1e-5 { + return uv_at(uvs, offset + k, p as usize); + } + } + offset += n as usize; + } + panic!("no refined face-vertex at the centre on the L"); + }; + let pinned = centre_uv_on_the_l(sdc::FVarLinearInterpolation::CornersPlus2); + assert!( + (pinned[0] - 0.5).abs() < 1e-5 && (pinned[1] - 0.5).abs() < 1e-5, + "cornersPlus2 keeps the concave corner at (0.5, 0.5), got {pinned:?}" + ); + let smoothed = centre_uv_on_the_l(sdc::FVarLinearInterpolation::CornersPlus1); + assert!( + (smoothed[0] - 0.5).abs() > 1e-3 || (smoothed[1] - 0.5).abs() > 1e-3, + "cornersPlus1 smooths the concave corner, got {smoothed:?}" + ); +} + +#[test] +fn vertex_chart_refines_like_the_points() { + let (points, counts, indices) = flat_quad(); + // Authored through `:indices`, reversed, so the indirection is + // exercised: point p reads values[3 - p]. + let values = [ + UNIT_SQUARE[3], + UNIT_SQUARE[2], + UNIT_SQUARE[1], + UNIT_SQUARE[0], + ]; + let req = SubdivRequest { + uvs: Some(UvChannel { + values: &values, + indices: Some(&[3, 2, 1, 0]), + face_varying: false, + linear: sdc::FVarLinearInterpolation::CornersPlus1, + }), + ..request(2) + }; + let out = subdivide(&points, &counts, &indices, &req).unwrap(); + let uvs = out.uvs.as_ref().unwrap(); + assert!(!uvs.face_varying && uvs.indices.is_none()); + assert_eq!(uvs.values.len(), out.points.len()); + assert_chart(&out, |p| [p.x / 2.0, p.y / 2.0], "vertex"); +} + +#[test] +fn chart_indices_are_checked_against_the_values() { + let chart = |indices: Option<&'static [i32]>| UvChannel { + values: &UNIT_SQUARE, + indices, + face_varying: true, + linear: sdc::FVarLinearInterpolation::All, + }; + assert!(chart(None).is_well_formed(4)); + assert!(!chart(None).is_well_formed(5), "too few values"); + assert!(chart(Some(&[0, 1, 2, 3])).is_well_formed(4)); + assert!( + !chart(Some(&[0, 1, 2, 4])).is_well_formed(4), + "out of range" + ); + assert!(!chart(Some(&[0, 1, -1, 3])).is_well_formed(4), "negative"); + assert!( + !chart(Some(&[0, 1, 2])).is_well_formed(4), + "too few indices" + ); +} + +/// The face-varying rule is one per refiner, and an authored chart's rule +/// wins — so the Ptex face table of a UV-charted mesh must come out +/// bit-identical under every rule, or its lookups would drift off their +/// sub-face. Five rules get there by sharing the refiner (the channel is +/// invariant under them); `none`, which smooths face-varying corners, +/// gets there through its own chartless refinement. +#[test] +fn ptex_channel_is_invariant_under_every_fvar_rule() { + let (points, counts, indices) = cube(); + let reference = subdivide( + &points, + &counts, + &indices, + &SubdivRequest { + want_face_uvs: true, + ..request(2) + }, + ) + .unwrap() + .faces + .unwrap(); + let chart: Vec<[f32; 2]> = (0..6).flat_map(|_| UNIT_SQUARE).collect(); + for linear in ALL_FVAR_RULES { + let req = SubdivRequest { + want_face_uvs: true, + uvs: Some(UvChannel { + values: &chart, + indices: None, + face_varying: true, + linear, + }), + ..request(2) + }; + let faces = subdivide(&points, &counts, &indices, &req) + .unwrap() + .faces + .unwrap(); + assert_eq!(faces.base_face, reference.base_face, "{linear:?}"); + let bits = |f: &SubdivFaces| -> Vec { + f.corner_uvs + .iter() + .flatten() + .flatten() + .map(|c| c.to_bits()) + .collect() + }; + assert_eq!(bits(&faces), bits(&reference), "{linear:?}"); + } +} + +fn quad_grid(n: usize) -> (Vec, Vec, Vec) { + let mut points = Vec::with_capacity((n + 1) * (n + 1)); + for j in 0..=n { + for i in 0..=n { + points.push(Vec3f::from([i as f32, j as f32, 0.0])); + } + } + let mut counts = Vec::with_capacity(n * n); + let mut indices = Vec::with_capacity(4 * n * n); + for j in 0..n { + for i in 0..n { + let v0 = (j * (n + 1) + i) as i32; + let v1 = v0 + 1; + let v2 = v1 + (n + 1) as i32; + let v3 = v0 + (n + 1) as i32; + counts.push(4); + indices.extend_from_slice(&[v0, v1, v2, v3]); + } + } + (points, counts, indices) +} + +/// What subdivision costs in memory, measured at the `subdivide()` +/// boundary: `transient` is the peak of live requested bytes while it +/// runs (dominated by the refiner, which retains every level 0..L — +/// a ×4/3 geometric series over the last level — plus the position +/// copies at the tail of the function), `resident` is what the returned +/// [`SubdividedMesh`] itself holds. The ceilings pin the per-face costs +/// so a regression (say, an accidentally retained per-level buffer) +/// fails loudly; they sit ~25% above the values measured at the time of +/// writing, printed by the table for recalibration. +/// +/// Ignored because the counters are process-global: run it alone — +/// `cargo test -p crust-core --lib subdivision_memory_probe -- --ignored --nocapture --test-threads=1` +#[test] +#[ignore = "allocation probe; run alone with --ignored --nocapture --test-threads=1"] +fn subdivision_memory_probe() { + use std::sync::atomic::Ordering::Relaxed; + + const GRID: usize = 64; // 4096 cage quads + let (points, counts, indices) = quad_grid(GRID); + // A continuous chart, shared across faces the way an unwrapped UV set + // is: one value per point, addressed per face-vertex. + let chart_values: Vec<[f32; 2]> = points.iter().map(|p| [p.x, p.y]).collect(); + let chart = UvChannel { + values: &chart_values, + indices: Some(&indices), + face_varying: true, + linear: sdc::FVarLinearInterpolation::Boundaries, + }; + + println!( + "\n{:>5} {:>5} {:>14} {:>14} {:>10} {:>10}", + "level", "uvs", "faces", "transient", "B/face", "resid B/f" + ); + for level in 1..=4u32 { + for mode in ["no", "ptex", "chart"] { + let want_uvs = mode == "ptex"; + let req = SubdivRequest { + want_face_uvs: want_uvs, + uvs: (mode == "chart").then_some(chart), + ..request(level) + }; + + let before = LIVE.load(Relaxed); + PEAK.store(before, Relaxed); + let out = subdivide(&points, &counts, &indices, &req).unwrap(); + let peak = PEAK.load(Relaxed); + let after = LIVE.load(Relaxed); + + let n_faces = out.counts.len(); + let transient = peak - before; + let resident = after - before; + let per_face = transient as f64 / n_faces as f64; + let res_per_face = resident as f64 / n_faces as f64; + println!( + "{:>5} {:>5} {:>14} {:>14} {:>10.1} {:>10.1}", + level, mode, n_faces, transient, per_face, res_per_face + ); + + assert_eq!(n_faces, counts.len() * 4usize.pow(level)); + // Ceilings only bind once the per-cage-face constants have + // amortized away; shallow levels are all fixed overhead. + if level >= 3 { + // Measured on a 64×64 cage: ~313 B/face with no + // face-varying channel, ~540 with the Ptex one, ~517 with + // a shared UV chart; resident 44.0 / 84.0 / 68.1 (the + // Ptex table's 40 B/face: four corner UVs plus an + // `Option` base face; the normals are 12 B per + // vertex since `compact-triangle-storage`, 16 before). + // Ceilings ~20% above those, per mode, so no mode can + // hide under another's. + let (ceiling, resident_ceiling) = match mode { + "no" => (380.0, 53.0), + "ptex" => (650.0, 100.0), + _ => (620.0, 82.0), + }; + assert!( + per_face < ceiling, + "transient {per_face:.1} B/face at level {level} \ + (uvs: {mode}) exceeds the {ceiling} B/face ceiling" + ); + assert!( + res_per_face < resident_ceiling, + "resident {res_per_face:.1} B/face at level {level} \ + (uvs: {mode}) exceeds the {resident_ceiling} B/face ceiling" + ); + } + drop(out); + } + } +} diff --git a/crates/crust-core/src/scene/subdiv/topology.rs b/crates/crust-core/src/scene/subdiv/topology.rs new file mode 100644 index 00000000..7ec240cf --- /dev/null +++ b/crates/crust-core/src/scene/subdiv/topology.rs @@ -0,0 +1,166 @@ +//! Cage validation and the topology the refiner is handed: crease runs, +//! corners, and the synthetic Ptex face-varying channel. + +use opensubdiv_rs::sdc; + +use super::SubdivError; + +/// The refiner indexes with `usize` counts and `u32` indices, and it cannot +/// skip a malformed face the way `triangulate` does — so the cage is checked +/// whole, up front. +pub(super) fn validate_cage( + n_verts: usize, + counts: &[i32], + indices: &[i32], +) -> Result<(Vec, Vec), SubdivError> { + let mut total = 0usize; + let mut counts_us = Vec::with_capacity(counts.len()); + for (face, &fc) in counts.iter().enumerate() { + if fc < 3 { + return Err(SubdivError::BadTopology(format!( + "face {face} has {fc} vertices (need at least 3)" + ))); + } + counts_us.push(fc as usize); + total += fc as usize; + } + if total != indices.len() { + return Err(SubdivError::BadTopology(format!( + "faceVertexCounts sums to {total} but faceVertexIndices has {} entries", + indices.len() + ))); + } + let mut indices_u32 = Vec::with_capacity(indices.len()); + for &i in indices { + if i < 0 || i as usize >= n_verts { + return Err(SubdivError::BadTopology(format!( + "face vertex index {i} out of range (mesh has {n_verts} points)" + ))); + } + indices_u32.push(i as u32); + } + Ok((counts_us, indices_u32)) +} + +/// Expands USD crease runs into the per-edge vertex pairs the refiner wants. +/// A run of `n` vertices contributes `n - 1` edges; `sharpnesses` carries +/// either one value per run or one per edge. Sharpness 10 is USD's "as sharp +/// as possible", which is exactly `sdc::SHARPNESS_INFINITE`; anything at or +/// above it is clamped there. +pub(super) fn expand_crease_runs( + indices: &[i32], + lengths: &[i32], + sharpnesses: &[f32], +) -> Result<(Vec<[u32; 2]>, Vec), SubdivError> { + if lengths.is_empty() { + return Ok((Vec::new(), Vec::new())); + } + let mut n_edges = 0usize; + let mut n_verts = 0usize; + for (run, &len) in lengths.iter().enumerate() { + if len < 2 { + return Err(SubdivError::BadTopology(format!( + "crease run {run} has length {len} (need at least 2 vertices)" + ))); + } + n_edges += len as usize - 1; + n_verts += len as usize; + } + if n_verts != indices.len() { + return Err(SubdivError::BadTopology(format!( + "creaseLengths sums to {n_verts} but creaseIndices has {} entries", + indices.len() + ))); + } + let per_run = sharpnesses.len() == lengths.len(); + if !per_run && sharpnesses.len() != n_edges { + return Err(SubdivError::BadTopology(format!( + "creaseSharpnesses has {} entries (want {} per-run or {n_edges} per-edge)", + sharpnesses.len(), + lengths.len() + ))); + } + let clamp = |s: f32| { + if s >= sdc::SHARPNESS_INFINITE { + sdc::SHARPNESS_INFINITE + } else { + s.max(0.0) + } + }; + let mut pairs = Vec::with_capacity(n_edges); + let mut weights = Vec::with_capacity(n_edges); + let mut off = 0usize; + let mut edge = 0usize; + for (run, &len) in lengths.iter().enumerate() { + for k in 0..len as usize - 1 { + let (a, b) = (indices[off + k], indices[off + k + 1]); + if a < 0 || b < 0 { + return Err(SubdivError::BadTopology(format!( + "crease run {run} has a negative vertex index" + ))); + } + pairs.push([a as u32, b as u32]); + weights.push(clamp(if per_run { + sharpnesses[run] + } else { + sharpnesses[edge] + })); + edge += 1; + } + off += len as usize; + } + Ok((pairs, weights)) +} + +pub(super) fn validate_corners( + indices: &[i32], + sharpnesses: &[f32], +) -> Result<(Vec, Vec), SubdivError> { + if indices.len() != sharpnesses.len() { + return Err(SubdivError::BadTopology(format!( + "cornerIndices has {} entries but cornerSharpnesses has {}", + indices.len(), + sharpnesses.len() + ))); + } + let mut out = Vec::with_capacity(indices.len()); + for &i in indices { + if i < 0 { + return Err(SubdivError::BadTopology( + "cornerIndices has a negative vertex index".into(), + )); + } + out.push(i as u32); + } + let weights = sharpnesses + .iter() + .map(|&s| { + if s >= sdc::SHARPNESS_INFINITE { + sdc::SHARPNESS_INFINITE + } else { + s.max(0.0) + } + }) + .collect(); + Ok((out, weights)) +} + +/// The synthetic face-varying channel carrying each cage face's Ptex +/// parameterization: one value per face-vertex (`0..sum(counts)` in authored +/// order, so no value is shared across faces), quads seeded with the four +/// Ptex corners. Non-quad faces get zeros — their descendants are marked +/// unmappable regardless, the channel just has to be well-formed. +pub(super) fn ptex_fvar_channel(counts: &[usize]) -> (Vec<[f32; 2]>, Vec) { + const QUAD: [[f32; 2]; 4] = [[0.0, 0.0], [1.0, 0.0], [1.0, 1.0], [0.0, 1.0]]; + let total: usize = counts.iter().sum(); + let mut values = Vec::with_capacity(total); + for &fc in counts { + if fc == 4 { + values.extend_from_slice(&QUAD); + } else { + values.extend(std::iter::repeat_n([0.0f32; 2], fc)); + } + } + let indices = (0..total as u32).collect(); + (values, indices) +} diff --git a/crates/crust-core/src/scene/subdiv/uniform.rs b/crates/crust-core/src/scene/subdiv/uniform.rs new file mode 100644 index 00000000..f1ebf382 --- /dev/null +++ b/crates/crust-core/src/scene/subdiv/uniform.rs @@ -0,0 +1,243 @@ +//! Uniform refinement: the cage refined to one level and snapped to the limit. + +use opensubdiv_rs::far::{ + FVarChannelDescriptor, PrimvarRefiner, TopologyDescriptor, TopologyRefinerFactory, + UniformOptions, +}; +use opensubdiv_rs::sdc; +use openusd::gf::Vec3f; + +use super::normals::smooth_normals; +use super::topology::{expand_crease_runs, ptex_fvar_channel, validate_cage, validate_corners}; +use super::{RefinedUvs, SubdivError, SubdivFaces, SubdivRequest, SubdivScheme, SubdividedMesh}; + +/// Uniformly refines the cage to `req.level` and snaps the result to the +/// limit surface. See the module docs for the shape of the answer. +pub(crate) fn subdivide( + points: &[Vec3f], + counts: &[i32], + indices: &[i32], + req: &SubdivRequest, +) -> Result { + // `none` is the one face-varying rule that smooths face-varying corners, + // which would slide the Ptex channel sharing its refiner off its + // sub-faces. Refine the face table on its own, chartless, and the rest + // without it — twice the refinement, for a material that reads Ptex *and* + // a `none` chart. (Pinned by `ptex_channel_is_invariant_under_every_fvar_rule`.) + if req.want_face_uvs + && req + .uvs + .is_some_and(|c| c.face_varying && c.linear == sdc::FVarLinearInterpolation::None) + { + let faces = subdivide( + points, + counts, + indices, + &SubdivRequest { uvs: None, ..*req }, + )? + .faces; + let mut out = subdivide( + points, + counts, + indices, + &SubdivRequest { + want_face_uvs: false, + ..*req + }, + )?; + out.faces = faces; + return Ok(out); + } + let (counts_us, indices_u32) = validate_cage(points.len(), counts, indices)?; + let (crease_pairs, crease_weights) = expand_crease_runs( + req.crease_indices, + req.crease_lengths, + req.crease_sharpnesses, + )?; + let corners = validate_corners(req.corner_indices, req.corner_sharpnesses)?; + + let scheme = match req.scheme { + SubdivScheme::CatmullClark => sdc::SchemeType::Catmark, + SubdivScheme::Bilinear => sdc::SchemeType::Bilinear, + SubdivScheme::Loop => sdc::SchemeType::Loop, + }; + // The face-varying rule is one per refiner, not per channel, so the + // authored chart's rule wins. The synthetic Ptex channel must refine + // bilinearly, and does under every rule but `none` (handled above) — + // each of its values is private to one face, so every one of its edges + // is a face-varying boundary, and its data is affine. Without a chart, + // `All`. + let chart = req.uvs.as_ref().filter(|c| c.face_varying); + let options = sdc::Options::default() + .with_vtx_boundary_interpolation(req.boundary) + .with_fvar_linear_interpolation( + chart.map_or(sdc::FVarLinearInterpolation::All, |c| c.linear), + ); + + // Loop cannot refine quads at all, and the Ptex channel only makes sense + // for the quad-split schemes. + let want_uvs = req.want_face_uvs && req.scheme != SubdivScheme::Loop; + let (fvar_uvs, fvar_indices) = if want_uvs { + ptex_fvar_channel(&counts_us) + } else { + (Vec::new(), Vec::new()) + }; + // A face-varying chart's per-face-vertex indices are its channel's + // topology; its values seed the refinement. + let chart_indices: Vec = match chart { + Some(c) => (0..indices.len()) + .map(|fv| c.value_index(fv) as u32) + .collect(), + None => Vec::new(), + }; + let mut channels = Vec::with_capacity(2); + if want_uvs { + channels.push(FVarChannelDescriptor::new(fvar_uvs.len(), &fvar_indices)); + } + let chart_channel = chart.map(|c| { + channels.push(FVarChannelDescriptor::new(c.values.len(), &chart_indices)); + channels.len() - 1 + }); + + let mut descriptor = TopologyDescriptor::new(points.len(), &counts_us, &indices_u32) + .with_creases(&crease_pairs, &crease_weights) + .with_corners(&corners.0, &corners.1); + if !channels.is_empty() { + descriptor = descriptor.with_fvar_channels(&channels); + } + + let mut refiner = + TopologyRefinerFactory::create(descriptor, scheme, options).map_err(SubdivError::Refine)?; + let level = req.level as usize; + refiner.refine_uniform(UniformOptions::new(level)); + + // Positions: interpolate level by level, then snap the last level to the + // limit surface. (For Bilinear the limit is the refined mesh itself; + // limit_level handles that uniformly.) + let primvar = PrimvarRefiner::new(&refiner); + let mut verts: Vec<[f32; 3]> = points.iter().map(|p| [p.x, p.y, p.z]).collect(); + for l in 1..=level { + let mut refined = vec![[0.0f32; 3]; refiner.level(l).num_vertices()]; + primvar.interpolate(l, &verts, &mut refined); + verts = refined; + } + let mut limit = vec![[0.0f32; 3]; verts.len()]; + primvar.limit(&verts, &mut limit); + + // Topology of the last level, back in the importer's array shapes. + let last = refiner.level(level); + let n_faces = last.num_faces(); + let mut out_counts = Vec::with_capacity(n_faces); + let mut out_indices = Vec::with_capacity(last.num_face_vertices_total()); + for f in 0..n_faces { + let fv = last.face_vertices(f); + out_counts.push(fv.len() as i32); + out_indices.extend(fv.iter().map(|&v| v as i32)); + } + + let faces = want_uvs.then(|| { + // Base-cage face per refined face: compose the one-step + // child-to-parent maps from the last refinement down to level 0. + let mut base_face: Vec = (0..n_faces as u32).collect(); + for l in (1..=level).rev() { + let refinement = refiner.refinement(l); + for f in &mut base_face { + *f = refinement.child_face_parent_face(*f as usize); + } + } + let base_face: Vec> = base_face + .into_iter() + .map(|f| (counts[f as usize] == 4).then_some(f)) + .collect(); + + // Sub-face corner UVs: refine the synthetic channel the same way the + // positions were refined, then read each face's four values. + let mut uvs = fvar_uvs.clone(); + for l in 1..=level { + let mut refined = vec![[0.0f32; 2]; refiner.level(l).num_fvar_values(0)]; + primvar.interpolate_face_varying(l, 0, &uvs, &mut refined); + uvs = refined; + } + let corner_uvs = (0..n_faces) + .map(|f| { + let fv = last.face_fvar_values(f, 0); + debug_assert_eq!(fv.len(), 4, "quad-split schemes only refine into quads"); + [ + uvs[fv[0] as usize], + uvs[fv[1] as usize], + uvs[fv[2] as usize], + uvs[fv[3] as usize], + ] + }) + .collect(); + SubdivFaces { + base_face, + corner_uvs, + } + }); + + let uvs = req.uvs.as_ref().map(|c| match chart_channel { + Some(ch) => { + // Face-varying: refine the values level by level, snap them to + // the limit, then read each refined face's entries. + let mut values = c.values.to_vec(); + for l in 1..=level { + let mut refined = vec![[0.0f32; 2]; refiner.level(l).num_fvar_values(ch)]; + primvar.interpolate_face_varying(l, ch, &values, &mut refined); + values = refined; + } + let mut limit_uvs = vec![[0.0f32; 2]; values.len()]; + primvar.limit_face_varying(ch, &values, &mut limit_uvs); + let mut fv_indices = Vec::with_capacity(out_indices.len()); + for f in 0..n_faces { + fv_indices.extend(last.face_fvar_values(f, ch).iter().map(|&v| v as i32)); + } + RefinedUvs { + values: limit_uvs, + indices: Some(fv_indices), + face_varying: true, + } + } + None => { + // Vertex: one value per point, refined and limited exactly like + // the positions. + let mut values: Vec<[f32; 2]> = (0..points.len()) + .map(|p| c.values[c.value_index(p)]) + .collect(); + for l in 1..=level { + let mut refined = vec![[0.0f32; 2]; refiner.level(l).num_vertices()]; + primvar.interpolate(l, &values, &mut refined); + values = refined; + } + let mut limit_uvs = vec![[0.0f32; 2]; values.len()]; + primvar.limit(&values, &mut limit_uvs); + RefinedUvs { + values: limit_uvs, + indices: None, + face_varying: false, + } + } + }); + + // Everything that needed the refiner is extracted; drop it — every + // level's topology, a ×4/3 of the last — before the result's own copies + // of the last level are made, so the two never coexist. This is the + // third of the transient the kernel design record costed. (`primvar` + // only borrows it; its last use is above.) + drop(refiner); + + let normals = smooth_normals(&limit, &out_counts, &out_indices); + let points: Vec = limit + .into_iter() + .map(|p| Vec3f::from([p[0], p[1], p[2]])) + .collect(); + + Ok(SubdividedMesh { + points, + counts: out_counts, + indices: out_indices, + normals, + faces, + uvs, + }) +} diff --git a/crates/crust-core/src/scene/usd_import/instancing.rs b/crates/crust-core/src/scene/usd_import/instancing.rs index 1f294158..54f40bc7 100644 --- a/crates/crust-core/src/scene/usd_import/instancing.rs +++ b/crates/crust-core/src/scene/usd_import/instancing.rs @@ -40,7 +40,7 @@ use super::attrs::{ custom_token, decode_i32_array, decode_i64_array, decode_vec3f_array, prim_ray_mask, value_at, }; use super::materials::{resolve_bound, resolve_material}; -use super::mesh::{MeshPlace, mesh_source, placement_scale}; +use super::mesh::{MeshNeeds, MeshPlace, mesh_source, placement_scale}; use super::shapes::{curve_segments, sphere_radius}; use super::xform::compose_with_parent; use super::{ImportCaches, WalkScope, prim_at, prune_reason}; @@ -200,10 +200,7 @@ pub(super) fn collect_proto_parts( if let Some(src) = mesh_source( &prim, &mesh, - material.face_texture().is_some(), - material.uses_uv(), - material.uv_primvar(), - displacement, + MeshNeeds::of(&*material, displacement), &mut caches.meshes.subdiv, part_world .as_ref() diff --git a/crates/crust-core/src/scene/usd_import/mesh.rs b/crates/crust-core/src/scene/usd_import/mesh.rs index 90bab1b2..74bba0c9 100644 --- a/crates/crust-core/src/scene/usd_import/mesh.rs +++ b/crates/crust-core/src/scene/usd_import/mesh.rs @@ -233,6 +233,22 @@ impl SubdivPolicy { } } + /// Counts a per-face tessellation into the load's statistics. + fn record_tessellation(&mut self, t: &subdiv::TessellatedMesh) { + self.per_face_meshes += 1; + for (into, from) in self.quality.iter_mut().zip(&t.quality) { + for (i, f) in into.iter_mut().zip(from) { + *i += f; + } + } + for (b, &n) in t.rate_bins.iter().enumerate() { + if self.rate_bins.len() <= b { + self.rate_bins.resize(b + 1, 0); + } + self.rate_bins[b] += n; + } + } + /// The level a subdivision mesh with this cage is refined to, read for /// `place`. fn level_for( @@ -563,15 +579,12 @@ pub(super) fn emit_mesh( displacement, } = bound; let displacement = prim_displacement(prim, displacement); - let want_faces = material.face_texture().is_some(); - let want_uvs = material.uses_uv(); + let needs = MeshNeeds::of(&*material, displacement.as_deref()); + let (want_faces, want_uvs) = (needs.faces, needs.uvs); let Some(src) = mesh_source( prim, mesh, - want_faces, - want_uvs, - material.uv_primvar(), - displacement.as_deref(), + needs, &mut meshes.subdiv, MeshPlace::World(&world_xf), ) else { @@ -987,17 +1000,19 @@ pub(super) enum RefinedFaces { /// /// `None` when the required attributes are missing (matching /// [`mesh_arrays`]); any subdivision problem warns and degrades to the cage. -#[allow(clippy::too_many_arguments)] pub(super) fn mesh_source( prim: &Prim, mesh: &UsdMesh, - want_faces: bool, - want_uvs: bool, - uv_primvar: Option<&str>, - displacement: Option<&Displacement>, + needs: MeshNeeds<'_>, policy: &mut SubdivPolicy, place: MeshPlace<'_>, ) -> Option { + let MeshNeeds { + faces: want_faces, + uvs: want_uvs, + uv_primvar, + displacement, + } = needs; // A displacement reads its own chart, whether or not the surface does. let want_faces = want_faces || displacement.is_some_and(Displacement::needs_ptex); let want_uvs = want_uvs || displacement.is_some_and(Displacement::needs_uv); @@ -1027,33 +1042,7 @@ pub(super) fn mesh_source( ); } - let mut usd_scheme = subdivision_scheme(mesh); - // A displaced `none` mesh is diced bilinearly, so the displacement has - // vertices to move: its faces keep their flat shape until displaced. - // Under `CRUST_SUBDIV=0` it stays the faceted cage, as everything does. - // - // Except when its displacement reads Ptex and it has a face that is not a - // quad: refinement splits a triangle or an n-gon into quads no Ptex face - // addresses here, so those children would read no map at all. Its cage - // keeps every triangle addressable, so it stays the cage. - if let Some(d) = displacement - && usd_scheme == SubdivisionScheme::None - && policy.enabled - { - if d.needs_ptex() && counts.iter().any(|&c| c != 4) { - if !policy.ptex_cage_warned { - policy.ptex_cage_warned = true; - warn!( - "Mesh at {} (and possibly others): subdivisionScheme = none with a Ptex \ - displacement and non-quad faces is displaced at its cage — refining it \ - would leave the children of its triangles with no Ptex face", - prim.path() - ); - } - } else { - usd_scheme = SubdivisionScheme::Bilinear; - } - } + let usd_scheme = effective_scheme(prim, mesh, &counts, displacement, policy); if !policy.enabled || usd_scheme == SubdivisionScheme::None { return Some(cage(points, counts, indices, uvs)); } @@ -1122,73 +1111,20 @@ pub(super) fn mesh_source( SubdivisionScheme::None => return Some(cage(points, counts, indices, uvs)), }; - let boundary = match mesh - .interpolate_boundary_attr() - .get_at::(eval_time()) - .ok() - .flatten() - .unwrap_or_default() - { - InterpolateBoundary::None => opensubdiv_rs::sdc::VtxBoundaryInterpolation::None, - InterpolateBoundary::EdgeOnly => opensubdiv_rs::sdc::VtxBoundaryInterpolation::EdgeOnly, - InterpolateBoundary::EdgeAndCorner => { - opensubdiv_rs::sdc::VtxBoundaryInterpolation::EdgeAndCorner - } - }; + let boundary = vtx_boundary(mesh); + let sharpness = SharpnessArrays::read(mesh); - let int_array = |attr| { - value_at(&attr) - .and_then(decode_i32_array) - .unwrap_or_default() - }; - let float_array = |attr| { - value_at(&attr) - .and_then(decode_f32_array) - .unwrap_or_default() - }; - let crease_indices = int_array(mesh.crease_indices_attr()); - let crease_lengths = int_array(mesh.crease_lengths_attr()); - let crease_sharpnesses = float_array(mesh.crease_sharpnesses_attr()); - let corner_indices = int_array(mesh.corner_indices_attr()); - let corner_sharpnesses = float_array(mesh.corner_sharpnesses_attr()); - - // The chart the refiner carries. One it cannot index (a negative or - // out-of-range entry) is dropped rather than refined into garbage: the - // surface then renders on the material's constant inputs, as every - // subdivided mesh did before charts were refined. - let chart = uvs.as_ref().and_then(|uv| { - let channel = subdiv::UvChannel { - values: &uv.values, - indices: uv.indices.as_deref(), - face_varying: uv.face_varying, - linear: face_varying_linear(mesh), - }; - let n_entries = if uv.face_varying { - indices.len() - } else { - points.len() - }; - if channel.is_well_formed(n_entries) { - Some(channel) - } else { - warn!( - "Mesh at {}: texture coordinates do not index cleanly into their \ - values — the subdivided surface renders without them", - prim.path() - ); - None - } - }); + let chart = refinable_chart(prim, mesh, uvs.as_ref(), points.len(), indices.len()); let req = subdiv::SubdivRequest { scheme, level, boundary, - crease_indices: &crease_indices, - crease_lengths: &crease_lengths, - crease_sharpnesses: &crease_sharpnesses, - corner_indices: &corner_indices, - corner_sharpnesses: &corner_sharpnesses, + crease_indices: &sharpness.crease_indices, + crease_lengths: &sharpness.crease_lengths, + crease_sharpnesses: &sharpness.crease_sharpnesses, + corner_indices: &sharpness.corner_indices, + corner_sharpnesses: &sharpness.corner_sharpnesses, want_face_uvs: want_faces, uvs: chart, }; @@ -1199,18 +1135,7 @@ pub(super) fn mesh_source( let segment = |pts: &[[f32; 3]]| rate.segment_culled(xf, pts, cull); match subdiv::tessellate_adaptive(&points, &counts, &indices, &req, rate.max, &segment) { Ok(t) => { - policy.per_face_meshes += 1; - for (into, from) in policy.quality.iter_mut().zip(&t.quality) { - for (i, f) in into.iter_mut().zip(from) { - *i += f; - } - } - for (b, &n) in t.rate_bins.iter().enumerate() { - if policy.rate_bins.len() <= b { - policy.rate_bins.resize(b + 1, 0); - } - policy.rate_bins[b] += n; - } + policy.record_tessellation(&t); debug!( "Mesh at {}: tessellated per face ({} Ptex faces -> {} triangles, edge rates {}..={})", prim.path(), @@ -1219,29 +1144,7 @@ pub(super) fn mesh_source( t.rate_range.0, t.rate_range.1 ); - let n_tris = t.indices.len() / 3; - return Some(MeshSource { - points: t.points, - counts: vec![3; n_tris], - indices: t.indices, - normals: Some(t.normals), - subdiv_faces: t.faces.map(RefinedFaces::PerFace), - base_face_count, - refined: true, - uvs: match (t.uvs, t.face_varying_uvs) { - (Some(values), _) => Some(UvSource { - values, - indices: None, - face_varying: false, - }), - (None, Some((values, corners))) => Some(UvSource { - values, - indices: Some(corners), - face_varying: true, - }), - (None, None) => None, - }, - }); + return Some(MeshSource::tessellated(t, base_face_count)); } Err(e) => { warn!( @@ -1260,20 +1163,7 @@ pub(super) fn mesh_source( base_face_count, refined.counts.len() ); - Some(MeshSource { - points: refined.points, - counts: refined.counts, - indices: refined.indices, - normals: Some(refined.normals), - subdiv_faces: refined.faces.map(RefinedFaces::Uniform), - base_face_count, - refined: true, - uvs: refined.uvs.map(|uv| UvSource { - values: uv.values, - indices: uv.indices, - face_varying: uv.face_varying, - }), - }) + Some(MeshSource::subdivided(refined, base_face_count)) } Err(e) => { warn!( @@ -1287,6 +1177,201 @@ pub(super) fn mesh_source( } } +/// What a mesh's material asks of its geometry beyond positions. +#[derive(Clone, Copy, Default)] +pub(super) struct MeshNeeds<'a> { + /// A per-face (Ptex) texture: keep the cage's face ids through refinement. + pub(super) faces: bool, + /// A UV texture: read, and refine with the surface, the texture chart. + pub(super) uvs: bool, + /// The chart's primvar, when the material's network names one + /// ([`Material::uv_primvar`]). + pub(super) uv_primvar: Option<&'a str>, + /// The displacement the prim applies, which reads its own chart. + pub(super) displacement: Option<&'a Displacement>, +} + +impl<'a> MeshNeeds<'a> { + /// What `material` reads, with the displacement the prim applies. + pub(super) fn of(material: &'a dyn Material, displacement: Option<&'a Displacement>) -> Self { + MeshNeeds { + faces: material.face_texture().is_some(), + uvs: material.uses_uv(), + uv_primvar: material.uv_primvar(), + displacement, + } + } +} + +/// The scheme a mesh is refined under: its `subdivisionScheme`, except that a +/// displaced `none` mesh is diced bilinearly, so the displacement has vertices +/// to move — its faces keep their flat shape until displaced. Under +/// `CRUST_SUBDIV=0` it stays the faceted cage, as everything does. +/// +/// Except when its displacement reads Ptex and it has a face that is not a +/// quad: refinement splits a triangle or an n-gon into quads no Ptex face +/// addresses here, so those children would read no map at all. Its cage keeps +/// every triangle addressable, so it stays the cage. +fn effective_scheme( + prim: &Prim, + mesh: &UsdMesh, + counts: &[i32], + displacement: Option<&Displacement>, + policy: &mut SubdivPolicy, +) -> SubdivisionScheme { + let scheme = subdivision_scheme(mesh); + let Some(d) = displacement else { + return scheme; + }; + if scheme != SubdivisionScheme::None || !policy.enabled { + return scheme; + } + if d.needs_ptex() && counts.iter().any(|&c| c != 4) { + if !policy.ptex_cage_warned { + policy.ptex_cage_warned = true; + warn!( + "Mesh at {} (and possibly others): subdivisionScheme = none with a Ptex \ + displacement and non-quad faces is displaced at its cage — refining it \ + would leave the children of its triangles with no Ptex face", + prim.path() + ); + } + scheme + } else { + SubdivisionScheme::Bilinear + } +} + +/// The mesh's `interpolateBoundary`, mapped one to one. +fn vtx_boundary(mesh: &UsdMesh) -> opensubdiv_rs::sdc::VtxBoundaryInterpolation { + use opensubdiv_rs::sdc::VtxBoundaryInterpolation as B; + match mesh + .interpolate_boundary_attr() + .get_at::(eval_time()) + .ok() + .flatten() + .unwrap_or_default() + { + InterpolateBoundary::None => B::None, + InterpolateBoundary::EdgeOnly => B::EdgeOnly, + InterpolateBoundary::EdgeAndCorner => B::EdgeAndCorner, + } +} + +/// A mesh's authored creases and corners, as the refiner reads them; an +/// unauthored array is empty. +struct SharpnessArrays { + crease_indices: Vec, + crease_lengths: Vec, + crease_sharpnesses: Vec, + corner_indices: Vec, + corner_sharpnesses: Vec, +} + +impl SharpnessArrays { + fn read(mesh: &UsdMesh) -> Self { + let ints = |attr| { + value_at(&attr) + .and_then(decode_i32_array) + .unwrap_or_default() + }; + let floats = |attr| { + value_at(&attr) + .and_then(decode_f32_array) + .unwrap_or_default() + }; + SharpnessArrays { + crease_indices: ints(mesh.crease_indices_attr()), + crease_lengths: ints(mesh.crease_lengths_attr()), + crease_sharpnesses: floats(mesh.crease_sharpnesses_attr()), + corner_indices: ints(mesh.corner_indices_attr()), + corner_sharpnesses: floats(mesh.corner_sharpnesses_attr()), + } + } +} + +/// The chart the refiner carries. One it cannot index (a negative or +/// out-of-range entry) is dropped rather than refined into garbage: the surface +/// then renders on the material's constant inputs, as every subdivided mesh did +/// before charts were refined. +fn refinable_chart<'a>( + prim: &Prim, + mesh: &UsdMesh, + uvs: Option<&'a UvSource>, + n_points: usize, + n_face_vertices: usize, +) -> Option> { + let uv = uvs?; + let channel = subdiv::UvChannel { + values: &uv.values, + indices: uv.indices.as_deref(), + face_varying: uv.face_varying, + linear: face_varying_linear(mesh), + }; + let n_entries = if uv.face_varying { + n_face_vertices + } else { + n_points + }; + if channel.is_well_formed(n_entries) { + Some(channel) + } else { + warn!( + "Mesh at {}: texture coordinates do not index cleanly into their \ + values — the subdivided surface renders without them", + prim.path() + ); + None + } +} + +impl MeshSource { + /// A per-face tessellation, all triangles. + fn tessellated(t: subdiv::TessellatedMesh, base_face_count: usize) -> Self { + let n_tris = t.indices.len() / 3; + MeshSource { + points: t.points, + counts: vec![3; n_tris], + indices: t.indices, + normals: Some(t.normals), + subdiv_faces: t.faces.map(RefinedFaces::PerFace), + base_face_count, + refined: true, + uvs: match (t.uvs, t.face_varying_uvs) { + (Some(values), _) => Some(UvSource { + values, + indices: None, + face_varying: false, + }), + (None, Some((values, corners))) => Some(UvSource { + values, + indices: Some(corners), + face_varying: true, + }), + (None, None) => None, + }, + } + } + + /// A uniform refinement. + fn subdivided(refined: subdiv::SubdividedMesh, base_face_count: usize) -> Self { + MeshSource { + points: refined.points, + counts: refined.counts, + indices: refined.indices, + normals: Some(refined.normals), + subdiv_faces: refined.faces.map(RefinedFaces::Uniform), + base_face_count, + refined: true, + uvs: refined.uvs.map(|uv| UvSource { + values: uv.values, + indices: uv.indices, + face_varying: uv.face_varying, + }), + } + } +} + /// The material's displacement as `prim` applies it: a /// `crust:displacementBound` authored on the mesh prim overrides the /// material's, since the same look can be bound to meshes of very different @@ -2083,10 +2168,7 @@ mod subdiv_policy_tests { let src = mesh_source( &prim, &mesh, - false, - false, - None, - None, + MeshNeeds::default(), &mut policy, MeshPlace::World(&GMat4::IDENTITY), ) @@ -2143,10 +2225,12 @@ mod subdiv_policy_tests { let src = mesh_source( &prim, &mesh, - false, - false, - None, - Some(&d), + MeshNeeds { + faces: false, + uvs: false, + uv_primvar: None, + displacement: Some(&d), + }, &mut policy, MeshPlace::World(&eye), ) @@ -2256,10 +2340,12 @@ mod displacement_tests { let src = mesh_source( &prim, &mesh, - false, - false, - None, - Some(d), + MeshNeeds { + faces: false, + uvs: false, + uv_primvar: None, + displacement: Some(d), + }, &mut arena.subdiv, MeshPlace::World(&eye), ) @@ -2466,10 +2552,12 @@ def Mesh "G" let src = mesh_source( &prim, &mesh, - false, - false, - None, - Some(&d), + MeshNeeds { + faces: false, + uvs: false, + uv_primvar: None, + displacement: Some(&d), + }, &mut arena.subdiv, MeshPlace::World(&GMat4::IDENTITY), ) @@ -2516,10 +2604,12 @@ def Mesh "T" let src = mesh_source( &prim, &mesh, - false, - false, - None, - Some(&d), + MeshNeeds { + faces: false, + uvs: false, + uv_primvar: None, + displacement: Some(&d), + }, &mut arena.subdiv, MeshPlace::World(&GMat4::IDENTITY), ) @@ -2536,10 +2626,12 @@ def Mesh "T" let src = mesh_source( &prim, &mesh, - false, - false, - None, - Some(&uv), + MeshNeeds { + faces: false, + uvs: false, + uv_primvar: None, + displacement: Some(&uv), + }, &mut arena.subdiv, MeshPlace::World(&GMat4::IDENTITY), ) diff --git a/crates/crust-core/src/scene/usd_import/mod.rs b/crates/crust-core/src/scene/usd_import/mod.rs index 11c3d206..1c4421f9 100644 --- a/crates/crust-core/src/scene/usd_import/mod.rs +++ b/crates/crust-core/src/scene/usd_import/mod.rs @@ -43,7 +43,7 @@ use crate::light::LightList; use crate::rt_world::WorldBuilder; use crate::scene::AssetLoader; use crate::scene::Scene; -use crate::stats::{ImageCounters, MemorySample, RenderStats, SceneCounters, SubdivisionCounters}; +use crate::stats::{MemorySample, RenderStats, SceneCounters, SubdivisionCounters}; use crate::tracer::RenderSettings; use crate::volume::VolumeRegion; @@ -567,36 +567,206 @@ fn open_stage(path: &Path, path_str: &str, mask: Option) -> Result, + camera: Option, + working: Option, +} + +impl ValidOptions { + fn check(options: &crate::UsdImportOptions) -> Result { + let time = options.frame; + // The authoritative check: every host reaches the importer through here, + // and nothing past this point expects a non-finite time. + if let Some(t) = time + && !t.is_finite() + { + return Err(crate::Error::InvalidFrame(t)); + } + // Same for the camera: a malformed path is refused before the stage is + // opened, not discovered after a four-minute traversal. + let requested_camera = match &options.camera { + Some(c) => Some( + sdf::path(c) + .ok() + .filter(|p| p.is_abs() && p.is_prim_path() && !p.is_abs_root()) + .ok_or_else(|| crate::Error::InvalidCameraPath(c.clone()))?, + ), + None => None, + }; + // A working space the host names is refused here, before the stage is + // opened, like a bad camera path. + let host_working = match &options.working_space { + Some(name) => Some(crate::color::working_space(name)?), + None => None, + }; + Ok(ValidOptions { + time, + camera: requested_camera, + working: host_working, + }) + } +} + +/// Walks the stage into `ctx`: in one pass when it is small or flat, else one +/// masked stage per top-level subtree (`chunks`), each dropped before the next +/// is composed — the memory bound the streaming import exists for. +fn traverse_stage( + path: &Path, + path_str: &str, + chunks: &[sdf::Path], + skip_stage_teardown: bool, + ctx: &mut ImportCtx, +) -> Result<(), crate::Error> { + if chunks.is_empty() { + // Small or flat stage: one pass, exactly as before. + debug!( + "Single-stage import (fewer than {MIN_STREAM_CHUNKS} subtrees, or \ + CRUST_STREAM_IMPORT=0)" + ); + let stage = open_stage(path, path_str, None)?; + if ctx.caches.meshes.subdiv.adaptive.is_some() { + count_placements(&stage, &mut ctx.caches); + } + traverse_into( + &stage, + prim_at(&stage, sdf::Path::abs_root()), + GMat4::IDENTITY, + ctx, + ); + release_stage(stage, skip_stage_teardown); + } else { + debug!("Streaming import over {} subtrees", chunks.len()); + for (n, chunk) in chunks.iter().enumerate() { + // Per chunk rather than per prim: this is the loop whose memory + // high-water mark the streaming import exists to bound, so the + // running totals are what say whether it is doing its job. + let chunk_start = Instant::now(); + debug!("Chunk {}/{}: {}", n + 1, chunks.len(), chunk); + let stage = open_stage(path, path_str, Some(chunk.clone()))?; + if ctx.caches.meshes.subdiv.adaptive.is_some() { + count_placements(&stage, &mut ctx.caches); + } + // Traverse from the root, not from `chunk`: a mask keeps the + // masked path's *ancestors* populated, so starting at the + // root picks up their transforms exactly as a full traversal + // would, while everything outside the chunk stays absent. + traverse_into( + &stage, + prim_at(&stage, sdf::Path::abs_root()), + GMat4::IDENTITY, + ctx, + ); + // Always dropped, the last chunk included: that is the memory + // bound streaming exists for, and a streamed import's peak often + // comes *after* the traversal — at the top-level BVH commit, on + // the island — where a kept stage would stack on top of it. + release_stage(stage, false); + // Separate this stage's prototypes from the next stage's — + // see ImportCaches::epoch. Deliberately not a clear: the + // mesh cache keys materials by Arc address, so nothing may + // be freed while it is live. + ctx.caches.epoch += 1; + ctx.caches.materials.epoch = ctx.caches.epoch; + debug!( + "Chunk {}/{} done in {:?} — running totals: {} geometries, {} light(s), \ + {} volume region(s), {} mesh placement(s) pending", + n + 1, + chunks.len(), + chunk_start.elapsed(), + ctx.world.count(), + ctx.lights.count(), + ctx.volumes.len(), + ctx.pending_meshes.len() + ); + } + } + Ok(()) +} + +/// The camera the traversal settled on: the one asked for, else — when the +/// stage's `RenderSettings.camera` names one that is not there — the first +/// camera met, else the procedural fallback's. A camera the host asked for by +/// path and the stage does not have is an error naming the alternatives. +fn resolve_camera(ctx: &mut ImportCtx) -> Result { + let camera = match ( + ctx.camera.take(), + ctx.wanted_camera.take(), + ctx.first_camera.take(), + ) { + (Some(c), _, _) => c, + (None, Some(CameraChoice::Requested(p)), _) => { + return Err(crate::Error::CameraNotFound { + path: p.to_string(), + available: ctx.cameras_seen.iter().map(ToString::to_string).collect(), + }); + } + (None, Some(CameraChoice::Settings(p)), Some((c, first))) => { + warn!( + "RenderSettings.camera targets {p}, which is not a camera on this stage — \ + rendering through {first} instead" + ); + c + } + (None, _, _) => { + warn!("USD stage has no UsdGeomCamera — falling back to world::get_settings camera"); + crate::world::get_settings().0 + } + }; + Ok(camera) +} + +/// The subdivision and displacement counters the traversal accumulated, into +/// `--stats`. +fn record_geometry_counters(meshes: &MeshArena, camera: &Camera, stats: &mut RenderStats) { + if meshes.subdiv.adaptive.is_some() { + let [interior, stitched] = meshes.subdiv.quality; + debug!( + "Per-face triangle shapes (4√3·area / Σ edge², bins ≥0.9 · ≥0.5 · ≥0.1 · ≥0.01 · \ + <0.01): interior {interior:?}, stitched {stitched:?}" + ); + } + if let Some(rate) = meshes.subdiv.adaptive { + // Both reads derive from one `CameraFrame`, but adaptive subdivision + // may read it off a different stage (the unloaded index while + // streaming); the two must still agree, or every level was chosen for + // a viewpoint the render does not use. + debug_assert!( + (Vec3::from(camera.origin()) - rate.eye).length() <= 1e-4 * rate.eye.length().max(1.0), + "adaptive subdivision read the camera at {:?}, the render camera is at {:?}", + rate.eye, + camera.origin() + ); + stats.subdivision = SubdivisionCounters { + adaptive: Some((rate.target, rate.max)), + levels: meshes.subdiv.levels.clone(), + shared_meshes: meshes.subdiv.shared_meshes, + shared_level: meshes.subdiv.level, + per_face_meshes: meshes.subdiv.per_face_meshes, + per_face_fallbacks: meshes.subdiv.per_face_fallbacks, + rate_bins: meshes.subdiv.rate_bins.clone(), + }; + } + + stats.displacement = crate::stats::DisplacementCounters { + frustum_skipped: meshes.subdiv.frustum_skipped, + ..meshes.displaced.clone() + }; +} + pub(crate) fn load_scene( path: &Path, assets: &dyn AssetLoader, options: &crate::UsdImportOptions, ) -> Result { - let time = options.frame; - // The authoritative check: every host reaches the importer through here, - // and nothing past this point expects a non-finite time. - if let Some(t) = time - && !t.is_finite() - { - return Err(crate::Error::InvalidFrame(t)); - } - // Same for the camera: a malformed path is refused before the stage is - // opened, not discovered after a four-minute traversal. - let requested_camera = match &options.camera { - Some(c) => Some( - sdf::path(c) - .ok() - .filter(|p| p.is_abs() && p.is_prim_path() && !p.is_abs_root()) - .ok_or_else(|| crate::Error::InvalidCameraPath(c.clone()))?, - ), - None => None, - }; - // A working space the host names is refused here, before the stage is - // opened, like a bad camera path. - let host_working = match &options.working_space { - Some(name) => Some(crate::color::working_space(name)?), - None => None, - }; + let ValidOptions { + time, + camera: requested_camera, + working: host_working, + } = ValidOptions::check(options)?; let _time_scope = EvalTimeScope::enter(time); let import_start = Instant::now(); let mut stats = RenderStats::new(); @@ -722,69 +892,13 @@ pub(crate) fn load_scene( }; let traverse_start = Instant::now(); - if chunks.is_empty() { - // Small or flat stage: one pass, exactly as before. - debug!( - "Single-stage import (fewer than {MIN_STREAM_CHUNKS} subtrees, or \ - CRUST_STREAM_IMPORT=0)" - ); - let stage = open_stage(path, path_str, None)?; - if ctx.caches.meshes.subdiv.adaptive.is_some() { - count_placements(&stage, &mut ctx.caches); - } - traverse_into( - &stage, - prim_at(&stage, sdf::Path::abs_root()), - GMat4::IDENTITY, - &mut ctx, - ); - release_stage(stage, options.skip_stage_teardown); - } else { - debug!("Streaming import over {} subtrees", chunks.len()); - for (n, chunk) in chunks.iter().enumerate() { - // Per chunk rather than per prim: this is the loop whose memory - // high-water mark the streaming import exists to bound, so the - // running totals are what say whether it is doing its job. - let chunk_start = Instant::now(); - debug!("Chunk {}/{}: {}", n + 1, chunks.len(), chunk); - let stage = open_stage(path, path_str, Some(chunk.clone()))?; - if ctx.caches.meshes.subdiv.adaptive.is_some() { - count_placements(&stage, &mut ctx.caches); - } - // Traverse from the root, not from `chunk`: a mask keeps the - // masked path's *ancestors* populated, so starting at the - // root picks up their transforms exactly as a full traversal - // would, while everything outside the chunk stays absent. - traverse_into( - &stage, - prim_at(&stage, sdf::Path::abs_root()), - GMat4::IDENTITY, - &mut ctx, - ); - // Always dropped, the last chunk included: that is the memory - // bound streaming exists for, and a streamed import's peak often - // comes *after* the traversal — at the top-level BVH commit, on - // the island — where a kept stage would stack on top of it. - release_stage(stage, false); - // Separate this stage's prototypes from the next stage's — - // see ImportCaches::epoch. Deliberately not a clear: the - // mesh cache keys materials by Arc address, so nothing may - // be freed while it is live. - ctx.caches.epoch += 1; - ctx.caches.materials.epoch = ctx.caches.epoch; - debug!( - "Chunk {}/{} done in {:?} — running totals: {} geometries, {} light(s), \ - {} volume region(s), {} mesh placement(s) pending", - n + 1, - chunks.len(), - chunk_start.elapsed(), - ctx.world.count(), - ctx.lights.count(), - ctx.volumes.len(), - ctx.pending_meshes.len() - ); - } - } + traverse_stage( + path, + path_str, + &chunks, + options.skip_stage_teardown, + &mut ctx, + )?; // The traverse also builds each mesh's and prototype's kernel scene, // so its own BVH work is inside this figure; the separate "Commit @@ -804,30 +918,7 @@ pub(crate) fn load_scene( ctx.volumes.len() ); - let camera = match ( - ctx.camera.take(), - ctx.wanted_camera.take(), - ctx.first_camera.take(), - ) { - (Some(c), _, _) => c, - (None, Some(CameraChoice::Requested(p)), _) => { - return Err(crate::Error::CameraNotFound { - path: p.to_string(), - available: ctx.cameras_seen.iter().map(ToString::to_string).collect(), - }); - } - (None, Some(CameraChoice::Settings(p)), Some((c, first))) => { - warn!( - "RenderSettings.camera targets {p}, which is not a camera on this stage — \ - rendering through {first} instead" - ); - c - } - (None, _, _) => { - warn!("USD stage has no UsdGeomCamera — falling back to world::get_settings camera"); - crate::world::get_settings().0 - } - }; + let camera = resolve_camera(&mut ctx)?; // Every chunk has been walked, so each mesh's placement count is final // and the deferred instance-vs-bake decisions can be made. Must happen @@ -843,39 +934,7 @@ pub(crate) fn load_scene( ctx.lights.hide_infinite_from_camera(); } - if ctx.caches.meshes.subdiv.adaptive.is_some() { - let [interior, stitched] = ctx.caches.meshes.subdiv.quality; - debug!( - "Per-face triangle shapes (4√3·area / Σ edge², bins ≥0.9 · ≥0.5 · ≥0.1 · ≥0.01 · \ - <0.01): interior {interior:?}, stitched {stitched:?}" - ); - } - if let Some(rate) = ctx.caches.meshes.subdiv.adaptive { - // Both reads derive from one `CameraFrame`, but adaptive subdivision - // may read it off a different stage (the unloaded index while - // streaming); the two must still agree, or every level was chosen for - // a viewpoint the render does not use. - debug_assert!( - (Vec3::from(camera.origin()) - rate.eye).length() <= 1e-4 * rate.eye.length().max(1.0), - "adaptive subdivision read the camera at {:?}, the render camera is at {:?}", - rate.eye, - camera.origin() - ); - stats.subdivision = SubdivisionCounters { - adaptive: Some((rate.target, rate.max)), - levels: ctx.caches.meshes.subdiv.levels.clone(), - shared_meshes: ctx.caches.meshes.subdiv.shared_meshes, - shared_level: ctx.caches.meshes.subdiv.level, - per_face_meshes: ctx.caches.meshes.subdiv.per_face_meshes, - per_face_fallbacks: ctx.caches.meshes.subdiv.per_face_fallbacks, - rate_bins: ctx.caches.meshes.subdiv.rate_bins.clone(), - }; - } - - stats.displacement = crate::stats::DisplacementCounters { - frustum_skipped: ctx.caches.meshes.subdiv.frustum_skipped, - ..ctx.caches.meshes.displaced.clone() - }; + record_geometry_counters(&ctx.caches.meshes, &camera, &mut stats); let pending = std::mem::take(&mut ctx.pending_meshes); flush_meshes(&mut ctx.world, &mut ctx.caches.meshes, pending); @@ -910,13 +969,7 @@ pub(crate) fn load_scene( lights: ctx.lights.count(), volumes: ctx.volumes.len(), }; - let (w, h) = settings.get_dimensions(); - stats.image = ImageCounters { - width: w, - height: h, - samples_per_pixel: settings.samples_per_pixel(), - max_depth: settings.max_depth(), - }; + stats.image = (&settings).into(); let mut scene = Scene::new(camera, committed, ctx.lights, settings).with_volumes(ctx.volumes); scene.stats = stats; diff --git a/crates/crust-core/src/stats.rs b/crates/crust-core/src/stats.rs index 40086ecf..0d60f65c 100644 --- a/crates/crust-core/src/stats.rs +++ b/crates/crust-core/src/stats.rs @@ -754,23 +754,9 @@ pub(crate) fn human_duration(d: Duration) -> String { } } -impl fmt::Display for RenderStats { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - // Wide enough that the longest phase name ("Commit acceleration - // structure", nested one level) still clears the time column. - const WIDTH: usize = 84; - const NAME: usize = 36; - let rule = "-".repeat(WIDTH); - let total = self.total(); - let total_secs = total.as_secs_f64(); - let pct = |d: Duration| { - if total_secs > 0.0 { - 100.0 * d.as_secs_f64() / total_secs - } else { - 0.0 - } - }; - +impl RenderStats { + /// The header, the image and the scene inventory: geometry, materials, lights, subdivision, displacement, volumes, kernel memory. + fn write_scene(&self, f: &mut fmt::Formatter<'_>, rule: &str) -> fmt::Result { writeln!(f, "{rule}")?; writeln!(f, "Render Statistics")?; writeln!(f, "{rule}")?; @@ -975,7 +961,11 @@ impl fmt::Display for RenderStats { if let Some(peak) = peak_memory_bytes() { writeln!(f, " {:<28} {}", "peak memory (RSS)", human_bytes(peak))?; } + Ok(()) + } + /// `Ray Statistics`, when anything was traced. + fn write_rays(&self, f: &mut fmt::Formatter<'_>, rule: &str) -> fmt::Result { // -- Ray statistics ------------------------------------------- let r = &self.rays; if !r.is_empty() { @@ -1146,7 +1136,11 @@ impl fmt::Display for RenderStats { )?; } } + Ok(()) + } + /// The `.tx` / preloaded texture section, when any texture was used. + fn write_textures(&self, f: &mut fmt::Formatter<'_>, rule: &str) -> fmt::Result { // -- Textures -------------------------------------------------- let t = &self.textures; if !t.is_empty() { @@ -1259,7 +1253,11 @@ impl fmt::Display for RenderStats { writeln!(f, " {:<28} {}", "tile read errors", count(t.errors))?; } } + Ok(()) + } + /// The Ptex section, when any Ptex texture was used. + fn write_ptex(&self, f: &mut fmt::Formatter<'_>, rule: &str) -> fmt::Result { // -- Ptex ------------------------------------------------------ let p = &self.ptex; if !p.is_empty() { @@ -1404,11 +1402,21 @@ impl fmt::Display for RenderStats { )?; } } + Ok(()) + } - if self.phases.is_empty() { - return Ok(()); - } - + /// The phases, by execution tree and by time; percentages are of the top-level total. + fn write_phases(&self, f: &mut fmt::Formatter<'_>, rule: &str) -> fmt::Result { + const NAME: usize = REPORT_NAME; + let total = self.total(); + let total_secs = total.as_secs_f64(); + let pct = |d: Duration| { + if total_secs > 0.0 { + 100.0 * d.as_secs_f64() / total_secs + } else { + 0.0 + } + }; // -- Phases by execution tree ---------------------------------- writeln!(f, "{rule}")?; writeln!(f, "Phases by execution tree (wall clock)")?; @@ -1461,7 +1469,123 @@ impl fmt::Display for RenderStats { width = NAME - 3 )?; } + Ok(()) + } +} + +/// The kernel's BVH traversal counters, per camera ray, as the `--stats` +/// section a `traversal-stats` build adds: queries, nodes, leaves and packets +/// per tree level, then which top-level instances the descents went into. +/// +/// One string rather than lines, so the caller can emit it as a single event: +/// a `println!` per row would leave these lines out of `--log-file`, and one +/// event per row would stamp each of them with a timestamp the table has no +/// column for. +#[cfg(feature = "traversal-stats")] +pub fn traversal_report(world: &crate::World, camera_rays: u64) -> String { + use crate::rt::traversal_stats as ts; + use std::fmt::Write as _; + let rays = camera_rays.max(1) as f64; + let per = |n: u64| n as f64 / rays; + let rule = "-".repeat(REPORT_WIDTH); + let mut out = String::new(); + // Infallible: `write!` into a String only fails if the formatter + // does, and none of these arguments can. + let _ = write!(out, "\n{rule}\nBVH Traversal (per camera ray)\n{rule}"); + for (level, name) in [(0usize, "top-level"), (1, "instanced")] { + let (q, nodes, leaves, packets, scalars) = ts::read_level(level); + if q == 0 { + continue; + } + let _ = write!( + out, + "\n {name:<12} queries {:>8.2} nodes {:>9.2} leaves {:>8.2} packets {:>7.2} scalar {:>8.2}", + per(q), + per(nodes), + per(leaves), + per(packets), + per(scalars), + ); + } + // Which top-level instances the descents went into. A top level that + // culls well spreads them thinly; one that does not concentrates them + // on whatever geometry every ray's path overlaps. The importer's + // DEBUG lines give each instancer's `geom ids a..b` range, which is + // how an id here is traced back to a prim. + let descents = ts::top_level_descents(); + let total: u64 = descents.iter().map(|d| d.1).sum(); + if total > 0 { + let mut acc = 0u64; + let mut marks = vec![]; + for (i, d) in descents.iter().enumerate() { + acc += d.1; + for f in [0.5, 0.9, 0.99] { + if (acc as f64) >= f * total as f64 && !marks.iter().any(|&(g, _)| g == f) { + marks.push((f, i + 1)); + } + } + } + let _ = write!( + out, + "\n top-level instances entered: {} of them, {:.1} descents per camera ray \ + (closest-hit and shadow rays; the rows above count closest-hit only)", + descents.len(), + per(total) + ); + for (f, n) in marks { + let _ = write!( + out, + "\n {:.0}% of descents go to {n} instances", + f * 100.0 + ); + } + let top: Vec<_> = descents.iter().take(40).collect(); + let ids: std::collections::HashSet = top.iter().map(|d| d.0).collect(); + let info: std::collections::HashMap = world + .describe_instances(&ids) + .into_iter() + .map(|(id, b, n, shared)| (id, (b, n, shared))) + .collect(); + let _ = write!( + out, + "\n {:>9} {:>7} {:>9} {:>8} {:>9} bounds", + "geom_id", "share", "per ray", "prims", "shared by" + ); + for &&(id, n) in &top { + let (b, prims, shared) = info[&id]; + let _ = write!( + out, + "\n {id:>9} {:>6.2}% {:>9.2} {prims:>8} {shared:>9} [{:.0} {:.0} {:.0}]..[{:.0} {:.0} {:.0}]", + 100.0 * n as f64 / total as f64, + per(n), + b.minimum.x, + b.minimum.y, + b.minimum.z, + b.maximum.x, + b.maximum.y, + b.maximum.z, + ); + } + } + out +} +/// Report column widths. Wide enough that the longest phase name ("Commit +/// acceleration structure", nested one level) still clears the time column. +const REPORT_WIDTH: usize = 84; +const REPORT_NAME: usize = 36; + +impl fmt::Display for RenderStats { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + let rule = "-".repeat(REPORT_WIDTH); + self.write_scene(f, &rule)?; + self.write_rays(f, &rule)?; + self.write_textures(f, &rule)?; + self.write_ptex(f, &rule)?; + if self.phases.is_empty() { + return Ok(()); + } + self.write_phases(f, &rule)?; // -- Render profile (`--profile`) -------------------------------- // After the phases, because it zooms into one of them: every figure // below is thread time inside the Render row above. diff --git a/crates/crust-core/src/tracer/mod.rs b/crates/crust-core/src/tracer/mod.rs index 793c713f..57d6024e 100644 --- a/crates/crust-core/src/tracer/mod.rs +++ b/crates/crust-core/src/tracer/mod.rs @@ -18,7 +18,7 @@ pub(crate) use path::surface_visibility; mod route; mod settings; -use path::{K_CAMERA, K_TIME, ray_cones_enabled, trace_path}; +use path::{K_CAMERA, K_TIME, PathContext, ray_cones_enabled, trace_path}; pub use path::ray_color; pub use settings::{ @@ -886,6 +886,15 @@ impl Renderer { .pixel_span(self.settings.width, self.settings.height) }); + let path_cx = PathContext { + world: &self.world, + lights: &self.lights, + volumes: &self.volumes, + depth: self.settings.max_depth as i32, + strategy: self.settings.sampling_strategy, + indirect_clamp: self.settings.indirect_clamp, + guiding: gctx, + }; for sample in state.taken..target { let primary = profile::scope_if::(Section::GeneratePrimary); let root = PathSampler::new(i as i32, j as i32, cfg.seed as i32, sample as i32) @@ -928,14 +937,8 @@ impl Renderer { unit.rays.camera_rays += 1; let color = trace_path::( &r, - &self.world, - &self.lights, - &self.volumes, - self.settings.max_depth as i32, - self.settings.sampling_strategy, - self.settings.indirect_clamp, + &path_cx, root, - gctx, &mut unit.samples, scratch, &mut unit.rays, diff --git a/crates/crust-core/src/tracer/path.rs b/crates/crust-core/src/tracer/path.rs index 67cb1326..fb67d175 100644 --- a/crates/crust-core/src/tracer/path.rs +++ b/crates/crust-core/src/tracer/path.rs @@ -88,6 +88,31 @@ fn russian_roulette( Some(p_survive) } +/// What every path of a render traces against: the scene, and the render +/// settings the integrator reads per path. Fixed for the whole render (a +/// guided pass's field included), so the renderer builds one per pixel and +/// every sample borrows it. +/// +/// Consumed only by the inlined `trace_path`, which destructures it on entry. +/// Never hand `&PathContext` on to a function LLVM keeps out of line: its +/// address then escapes, the struct must stay in memory, and every field is +/// reloaded after every opaque call — passing it to `volume_nee` and +/// `mixed_hair_shadow` cost cornellbox 0.08% of its instructions, and passing +/// it by value 0.23% (callgrind, 2 spp). Pass those helpers the fields. +#[derive(Clone, Copy)] +pub(super) struct PathContext<'a> { + pub(super) world: &'a World, + pub(super) lights: &'a LightList, + pub(super) volumes: &'a Volumes, + /// The longest path, in vertices. + pub(super) depth: i32, + pub(super) strategy: SamplingStrategy, + /// The firefly clamp on indirect light, `None` when off. + pub(super) indirect_clamp: Option, + /// The guiding field and whether this pass trains it. + pub(super) guiding: Option<&'a GuidingContext<'a>>, +} + pub fn ray_color( r: &Ray, world: &World, @@ -102,20 +127,16 @@ pub fn ray_color( // The one-shot entry point (benches and tests), so a scratch per call is // the right trade — the renderer's own paths reuse one per work unit. let mut scratch = PathScratch::new(depth.max(0) as usize); - trace_path::( - r, + let cx = PathContext { world, lights, volumes, depth, strategy, - None, - sampler, - None, - &mut no_training, - &mut scratch, - &mut stats, - ) + indirect_clamp: None, + guiding: None, + }; + trace_path::(r, &cx, sampler, &mut no_training, &mut scratch, &mut stats) } /// Are texture-filtering ray cones on? `CRUST_RAY_CONES=0` forces every @@ -924,22 +945,24 @@ fn volume_nee( /// draw, no weight — so both instantiations return the same radiance, and /// with `AOV = false` every `if AOV` block compiles away, leaving the /// function the beauty-only render has always run. -#[allow(clippy::too_many_arguments)] #[inline(always)] pub(super) fn trace_path( r: &Ray, - world: &World, - lights: &LightList, - volumes: &Volumes, - depth: i32, - strategy: SamplingStrategy, - indirect_clamp: Option, + cx: &PathContext<'_>, sampler: PathSampler, - guiding: Option<&GuidingContext>, train_out: &mut Vec, scratch: &mut PathScratch, stats: &mut RayStats, ) -> Vec3A { + let PathContext { + world, + lights, + volumes, + depth, + strategy, + indirect_clamp, + guiding, + } = *cx; let training = guiding.is_some_and(|g| g.training); // The bounce subtree; each vertex derives its own domain off this by depth. let path = sampler.new_domain(K_PATH); diff --git a/crates/crust-mtlx/src/eval.rs b/crates/crust-mtlx/src/eval.rs deleted file mode 100644 index b40c53ba..00000000 --- a/crates/crust-mtlx/src/eval.rs +++ /dev/null @@ -1,2185 +0,0 @@ -//! The MaterialX pattern graph, compiled to a flat program and run per hit. -//! -//! A `.mtlx` look-dev graph is evaluated *per shading point* — its textures, -//! masks and blends are the whole point — so this cannot be folded to -//! constants at import. But it must also not be walked as a name-addressed DOM -//! inside the integrator: every lookup would be a string hash, and the teapot's -//! ceramic graph alone is ~50 nodes consulted several times per path vertex -//! (once to sample, again for each NEE and guide evaluation). -//! -//! So the graph is **compiled once** into a [`Program`]: a topologically -//! ordered `Vec` whose operands are slot *indices* into the values -//! computed so far. Evaluation is then a linear scan with no branching on -//! names, no allocation (the value stack is a thread-local scratch buffer -//! reused across calls), and no hash lookups. -//! -//! The compiler is also where cycles and unknown operators are dealt with: -//! a node that cannot be compiled becomes a constant, so an unsupported -//! MaterialX node degrades that one input to its default rather than failing -//! the material. - -use super::parse::{Doc, Input, Node, Source}; -use super::value::{Val, arity_of, convert_color, is_color_type}; -use crate::texture::TextureRef; -use glam::Vec3A; - -/// The shading point a [`Program`] is evaluated at. -#[derive(Clone, Copy)] -pub struct ShadeCtx { - /// `primvars:st`, unwrapped — the integer part selects a UDIM tile. - pub uv: (f32, f32), - /// Geometric (ray-facing) world-space normal. - pub normal: Vec3A, - /// World-space tangent along increasing `u`; `ZERO` when unknown, which - /// makes `normalmap` pass the geometric normal straight through rather - /// than build a frame out of noise. - pub tangent: Vec3A, - /// Direction from the viewer *to* the surface — MaterialX's - /// `viewdirection`, which is the incoming ray's direction, not its - /// negation. - pub view: Vec3A, - pub position: Vec3A, - /// Diameter of the shading point's texture footprint, in the same UV - /// units as `uv` — what [`crate::Texture::eval`] filters over. `0.0` asks - /// every texture in the graph to point-sample, which is what a host that - /// tracks no footprint should leave it at. - pub uv_width: f32, -} - -/// A componentwise binary operator. -#[derive(Clone, Copy, Debug)] -pub enum BinOp { - Add, - Sub, - Mul, - Div, - Pow, - Min, - Max, - /// `a` with `b` lanes, for `combine`-style plumbing. - Modulo, -} - -/// A componentwise unary operator. -#[derive(Clone, Copy, Debug)] -pub enum UnOp { - Abs, - Ln, - Exp, - Sin, - Cos, - Asin, - Acos, - Sqrt, - Sign, - Floor, - Ceil, - Normalize, -} - -/// One instruction. Operands are slot indices, always strictly less than the -/// instruction's own slot — the compiler emits in topological order, so a -/// single forward pass evaluates the whole program. -#[derive(Clone, Debug)] -pub enum Op { - Const(Val), - /// A UV texture lookup. `tex` is `None` when the host declined the file, - /// in which case `fallback` — the node's `default` input, or mid-grey — - /// stands in, which is what keeps an undecodable texture from blackening - /// a surface. - Texture { - tex: Option, - fallback: Val, - /// `uvtiling` / `uvoffset` from a `tiledimage`; identity for `image`. - scale: [f32; 2], - offset: [f32; 2], - arity: u8, - /// Where to look up relative to the shading point, in footprint - /// widths (see `shifted_uv`). Zero everywhere except the copies of - /// a subgraph `heighttonormal` differentiates. - shift: [f32; 2], - /// The slot holding an authored `texcoord` connection, when the - /// image has one other than the default chart; `None` reads the - /// shading point's `uv` (displaced by `shift`). A connected - /// coordinate carries any shift in its own `TexCoord` ops, so a - /// coordinate that does not depend on the chart is not displaced. - coord: Option, - }, - /// `primvars:st` as a `vector2`, displaced by `shift` footprint widths - /// exactly as [`Op::Texture`] is. - TexCoord { - shift: [f32; 2], - }, - /// World-space geometric normal. - Normal, - ViewDirection, - Position, - Unary { - op: UnOp, - a: u32, - }, - Binary { - op: BinOp, - a: u32, - b: u32, - }, - /// `bg·(1−m) + fg·m`, MaterialX's `mix`. - Mix { - fg: u32, - bg: u32, - m: u32, - }, - Clamp { - a: u32, - low: u32, - high: u32, - }, - /// `(in − pivot)·amount + pivot`. - Contrast { - a: u32, - amount: u32, - pivot: u32, - }, - /// Linear rescale from `[inlow, inhigh]` to `[outlow, outhigh]`. - Remap { - a: u32, - in_low: u32, - in_high: u32, - out_low: u32, - out_high: u32, - }, - /// `amount − in`. - Invert { - a: u32, - amount: u32, - }, - /// Reinterpret at a different lane count, broadcasting a scalar. - Convert { - a: u32, - arity: u8, - }, - /// One lane of a wider value. - Extract { - a: u32, - index: usize, - }, - Combine3 { - a: u32, - b: u32, - c: u32, - }, - Combine2 { - a: u32, - b: u32, - }, - DotProduct { - a: u32, - b: u32, - }, - /// `dot(in.rgb, coeffs)`, keeping a `color4`'s alpha. - Luminance { - a: u32, - coeffs: u32, - }, - /// Decodes a tangent-space normal map into a world-space normal. - NormalMap { - a: u32, - scale: u32, - }, - /// Gulbrandsen's artist-friendly metal parameterisation. `extinction` - /// selects which of the node's two outputs this slot holds. - ArtisticIor { - reflectivity: u32, - edge: u32, - extinction: bool, - }, - /// Hermite interpolation between two edges. - Smoothstep { - a: u32, - low: u32, - high: u32, - }, - /// MaterialX `hsvadjust`: to HSV, add `amount.x` to the hue and scale - /// saturation and value by `amount.y` / `amount.z`, and back. - HsvAdjust { - a: u32, - amount: u32, - }, - /// MaterialX `heighttonormal`, from the height at half a footprint either - /// side of the shading point along `u` (`xp`, `xm`) and `v` (`yp`, `ym`). - /// The result is encoded in `[0, 1]`, like a normal-map texel. - HeightToNormal { - xp: u32, - xm: u32, - yp: u32, - ym: u32, - scale: u32, - }, -} - -impl Op { - /// Calls `f` on every operand slot index, in a fixed order. - pub fn for_each_operand(&mut self, mut f: impl FnMut(&mut u32)) { - match self { - Op::Texture { coord, .. } => { - if let Some(c) = coord { - f(c); - } - } - Op::Const(_) | Op::TexCoord { .. } | Op::Normal | Op::ViewDirection | Op::Position => {} - Op::Unary { a, .. } | Op::Convert { a, .. } | Op::Extract { a, .. } => f(a), - Op::Binary { a, b, .. } - | Op::Invert { a, amount: b } - | Op::Combine2 { a, b } - | Op::Luminance { a, coeffs: b } - | Op::DotProduct { a, b } - | Op::NormalMap { a, scale: b } - | Op::HsvAdjust { a, amount: b } - | Op::ArtisticIor { - reflectivity: a, - edge: b, - .. - } => { - f(a); - f(b); - } - Op::Mix { fg, bg, m } => { - f(fg); - f(bg); - f(m); - } - Op::Clamp { a, low, high } | Op::Smoothstep { a, low, high } => { - f(a); - f(low); - f(high); - } - Op::Contrast { a, amount, pivot } => { - f(a); - f(amount); - f(pivot); - } - Op::Combine3 { a, b, c } => { - f(a); - f(b); - f(c); - } - Op::Remap { - a, - in_low, - in_high, - out_low, - out_high, - } => { - f(a); - f(in_low); - f(in_high); - f(out_low); - f(out_high); - } - Op::HeightToNormal { - xp, - xm, - yp, - ym, - scale, - } => { - f(xp); - f(xm); - f(yp); - f(ym); - f(scale); - } - } - } - - /// Whether the result depends on nothing but the operands — no shading - /// point, no texture. Such an op over constant operands is a constant. - fn is_pure(&self) -> bool { - match self { - Op::Texture { tex, .. } => tex.is_none(), - Op::TexCoord { .. } - | Op::Normal - | Op::ViewDirection - | Op::Position - | Op::NormalMap { .. } => false, - _ => true, - } - } -} - -/// A compiled pattern graph. -/// -/// Slots `0..consts.len()` hold `consts`, copied in once per evaluation; slot -/// `consts.len() + i` holds the value of `ops[i]`. The compiler leaves -/// `consts` empty and emits every literal as an [`Op::Const`]; -/// [`Program::optimize`] moves them (and everything computable from them) -/// into `consts`. -#[derive(Clone, Default)] -pub struct Program { - pub consts: Vec, - pub ops: Vec, -} - -impl Program { - /// Number of slots an evaluation fills. - pub fn len(&self) -> usize { - self.consts.len() + self.ops.len() - } - - pub fn is_empty(&self) -> bool { - self.len() == 0 - } - - /// Whether every operand refers to a slot strictly before its user's — - /// what the compiler always emits, and what anything that runs a program - /// by other means than [`Program::eval`]'s bounds-checked reads relies on. - pub fn is_well_formed(&self) -> bool { - let nc = self.consts.len(); - self.ops.iter().enumerate().all(|(i, op)| { - let mut ok = true; - op.clone() - .for_each_operand(|o| ok &= (*o as usize) < nc + i); - ok - }) - } - - /// One instruction's value, given the slots computed before it — the - /// interpreter's own step, for an evaluator that runs some ops another - /// way and hands the rest back here (crust-jit), so that both produce the - /// same bits by construction. - pub fn apply_op(op: &Op, slots: &[Val], ctx: &ShadeCtx) -> Val { - apply(op, slots, ctx) - } - - /// Evaluates every instruction into `slots`, which is resized as needed. - /// - /// The caller owns the buffer so it can be reused across shading calls — - /// a fresh `Vec` per call would allocate once per BSDF evaluation, which - /// at several evaluations per path vertex is the kind of cost that does - /// not show up in a profile as one hot line. - pub fn eval(&self, ctx: &ShadeCtx, slots: &mut Vec) { - slots.clear(); - slots.reserve(self.len()); - slots.extend_from_slice(&self.consts); - for op in &self.ops { - let v = apply(op, slots, ctx); - slots.push(v); - } - } - - /// Rewrites the program so it computes the same values at `roots` with - /// less work per evaluation, and returns where each old slot went - /// (`None` for a slot that no longer exists). - /// - /// Three passes, each exact rather than approximate: - /// - /// - **Constant folding.** An op whose operands are all constant, and - /// which reads neither the shading point nor a texture, is evaluated - /// here — by `apply`, the function the interpreter runs, so the value - /// is the one every hit would have computed, bit for bit. - /// - **Constant hoisting and deduplication.** Constants move into - /// [`Program::consts`], copied in with one `memcpy` per evaluation - /// instead of one dispatched instruction each, and bitwise-equal ones - /// share a slot (the compiler emits a fresh constant for every literal - /// and every unauthored input's default). - /// - **Dead-code elimination.** Ops nothing at `roots` depends on are - /// dropped. - /// - /// The surviving ops keep their relative order, so operands still precede - /// their users. - /// - /// A malformed program — an operand that does not precede its user, or a - /// root out of range — is returned unchanged with the identity remap: the - /// interpreter already reads such an operand as zero, and rewriting it - /// would have to invent a meaning for it. - pub fn optimize(&self, roots: &[u32]) -> (Program, Vec>) { - let n = self.len(); - let nc = self.consts.len(); - if !self.is_well_formed() || roots.iter().any(|&r| r as usize >= n) { - return (self.clone(), (0..n as u32).map(Some).collect()); - } - // Every slot's value, where it is a compile-time constant. - let mut known: Vec> = self.consts.iter().copied().map(Some).collect(); - known.resize(n, None); - let ctx = ShadeCtx { - uv: (0.0, 0.0), - normal: Vec3A::Z, - tangent: Vec3A::ZERO, - view: -Vec3A::Z, - position: Vec3A::ZERO, - uv_width: 0.0, - }; - let mut scratch: Vec = vec![Val::ZERO; n]; - for (i, op) in self.ops.iter().enumerate() { - let slot = nc + i; - let mut all_known = op.is_pure(); - op.clone() - .for_each_operand(|o| match known.get(*o as usize).copied().flatten() { - Some(v) => scratch[*o as usize] = v, - None => all_known = false, - }); - if all_known { - // `apply` reads operands by slot, so hand it the known values - // at their own indices. - known[slot] = Some(apply(op, &scratch[..slot], &ctx)); - } - } - - // Liveness, from the roots back. - let mut live = vec![false; n]; - for &r in roots { - if let Some(l) = live.get_mut(r as usize) { - *l = true; - } - } - for i in (0..self.ops.len()).rev() { - let slot = nc + i; - if live[slot] && known[slot].is_none() { - self.ops[i].clone().for_each_operand(|o| { - if let Some(l) = live.get_mut(*o as usize) { - *l = true; - } - }); - } - } - - // Constants first, deduplicated bitwise, then the surviving ops. - let mut out = Program::default(); - let mut remap: Vec> = vec![None; n]; - let key = |v: &Val| (v.v.map(f32::to_bits), v.arity); - let mut seen: std::collections::HashMap<([u32; 4], u8), u32> = Default::default(); - for slot in 0..n { - if let (true, Some(v)) = (live[slot], known[slot]) { - let idx = *seen.entry(key(&v)).or_insert_with(|| { - out.consts.push(v); - (out.consts.len() - 1) as u32 - }); - remap[slot] = Some(idx); - } - } - let base = out.consts.len(); - for (i, op) in self.ops.iter().enumerate() { - let slot = nc + i; - if live[slot] && known[slot].is_none() { - let mut op = op.clone(); - op.for_each_operand(|o| { - // A live op's operands are live, so they were placed. - *o = remap[*o as usize].unwrap_or(0); - }); - remap[slot] = Some((base + out.ops.len()) as u32); - out.ops.push(op); - } - } - (out, remap) - } -} - -/// One instruction's value, given the slots computed before it. -/// -/// Shared by [`Program::eval`] and the constant folder in -/// [`Program::optimize`], which is what makes a folded constant exactly the -/// value the interpreter would have produced. -#[inline(always)] -fn apply(op: &Op, slots: &[Val], ctx: &ShadeCtx) -> Val { - // Every operand index was emitted before this instruction, so the slot - // exists. `get` rather than indexing keeps a malformed program from - // panicking inside the integrator. - let g = |i: u32| -> Val { slots.get(i as usize).copied().unwrap_or(Val::ZERO) }; - match op { - Op::Const(v) => *v, - Op::Texture { - tex, - fallback, - scale, - offset, - arity, - shift, - coord, - } => match tex { - Some(t) => { - let (u, v) = match coord { - Some(c) => { - let c = g(*c); - // A `float` coordinate is both axes, as MaterialX's - // implicit promotion to `vector2` makes it. - if c.arity == 1 { - (c.x(), c.x()) - } else { - (c.v[0], c.v[1]) - } - } - None => shifted_uv(ctx, *shift), - }; - let u = u * scale[0] + offset[0]; - let v = v * scale[1] + offset[1]; - // `uvtiling` scales the coordinates, so it scales the - // footprint with them: a texture tiled 10× is being - // minified 10× and must read a coarser level to match. - // The two axes are averaged because the width is one - // isotropic number. - let w = ctx.uv_width * 0.5 * (scale[0].abs() + scale[1].abs()); - let rgba = t.eval(u, v, w); - Val { - v: rgba, - arity: *arity, - } - } - None => *fallback, - }, - Op::TexCoord { shift } => { - let (u, v) = shifted_uv(ctx, *shift); - Val::vec2(u, v) - } - Op::Normal => ctx.normal.into(), - Op::ViewDirection => ctx.view.into(), - Op::Position => ctx.position.into(), - Op::Unary { op, a } => { - let a = g(*a); - match op { - UnOp::Abs => a.map(f32::abs), - // Guarded so a zero or negative operand — which the - // teapot's Beer-Lambert chain can produce from a - // black texel — yields a finite value instead of an - // infinity that then poisons every downstream lane. - // The guard is OSL's `safe_log`, which the MaterialX - // reference runs: the operand is raised to the smallest - // normal float, so `ln(0)` is `ln(f32::MIN_POSITIVE)`. - UnOp::Ln => a.map(|x| x.max(f32::MIN_POSITIVE).ln()), - UnOp::Exp => a.map(|x| x.clamp(-88.0, 88.0).exp()), - UnOp::Sin => a.map(f32::sin), - UnOp::Cos => a.map(f32::cos), - UnOp::Asin => a.map(|x| x.clamp(-1.0, 1.0).asin()), - UnOp::Acos => a.map(|x| x.clamp(-1.0, 1.0).acos()), - UnOp::Sqrt => a.map(|x| x.max(0.0).sqrt()), - // Not `f32::signum`, which answers ±1 for ±0: MaterialX's - // `sign` of zero is zero. - UnOp::Sign => a.map(|x| { - if x > 0.0 { - 1.0 - } else if x < 0.0 { - -1.0 - } else { - x - } - }), - UnOp::Floor => a.map(f32::floor), - UnOp::Ceil => a.map(f32::ceil), - UnOp::Normalize => normalize(a), - } - } - Op::Binary { op, a, b } => { - let (a, b) = (g(*a), g(*b)); - match op { - BinOp::Add => a + b, - BinOp::Sub => a - b, - BinOp::Mul => a * b, - // A zero divisor is a real possibility in these - // graphs (`1 / transmittance` with a black channel), - // and an infinity survives every later multiply. - BinOp::Div => a.zip(b, |x, y| if y.abs() > 1e-20 { x / y } else { 0.0 }), - BinOp::Pow => a.zip(b, safe_pow), - BinOp::Min => a.zip(b, f32::min), - BinOp::Max => a.zip(b, f32::max), - BinOp::Modulo => a.zip(b, floored_mod), - } - } - // `bg·(1−m) + fg·m`. Written as two weighted terms rather - // than `bg + (fg−bg)·m` so that a `float` mix against wider - // operands broadcasts through `zip`'s promotion in both - // terms alike. - Op::Mix { fg, bg, m } => { - let (fg, bg, m) = (g(*fg), g(*bg), g(*m)); - let inv = m.map(|x| 1.0 - x); - bg * inv + fg * m - } - // `max(min(in, high), low)`, OSL's order: it only shows when - // `low > high`, where `low` wins. - Op::Clamp { a, low, high } => { - let (a, lo, hi) = (g(*a), g(*low), g(*high)); - a.zip(hi, f32::min).zip(lo, f32::max) - } - Op::Contrast { a, amount, pivot } => { - let (a, amt, piv) = (g(*a), g(*amount), g(*pivot)); - (a - piv) * amt + piv - } - Op::Remap { - a, - in_low, - in_high, - out_low, - out_high, - } => { - let (a, il, ih, ol, oh) = (g(*a), g(*in_low), g(*in_high), g(*out_low), g(*out_high)); - let t = (a - il).zip(ih - il, |x, d| if d.abs() > 1e-20 { x / d } else { 0.0 }); - ol + (oh - ol) * t - } - Op::Invert { a, amount } => g(*amount) - g(*a), - Op::Convert { a, arity } => convert(g(*a), *arity), - Op::Extract { a, index } => Val::float(g(*a).v[(*index).min(3)]), - Op::Combine3 { a, b, c } => Val::vec3(g(*a).x(), g(*b).x(), g(*c).x()), - Op::Combine2 { a, b } => combine2(g(*a), g(*b)), - Op::DotProduct { a, b } => dot(g(*a), g(*b)), - Op::Luminance { a, coeffs } => { - let a = g(*a); - let l = a.rgb().dot(g(*coeffs).rgb()); - // The input's width: `color3` is the grey `(l, l, l)`, and - // `color4` keeps its alpha. - if a.arity == 4 { - Val::vec4(l, l, l, a.v[3]) - } else { - Val::float(l).broadcast_to(a.arity) - } - } - Op::NormalMap { a, scale } => normal_map(g(*a), g(*scale), ctx).into(), - Op::ArtisticIor { - reflectivity, - edge, - extinction, - } => { - let (n, k) = artistic_ior(g(*reflectivity).rgb(), g(*edge).rgb()); - if *extinction { k.into() } else { n.into() } - } - Op::Smoothstep { a, low, high } => { - let (a, lo, hi) = (g(*a), g(*low), g(*high)); - let n = a.arity.max(lo.arity).max(hi.arity); - let (a, lo, hi) = (a.broadcast_to(n), lo.broadcast_to(n), hi.broadcast_to(n)); - Val { - v: [0, 1, 2, 3].map(|i| smoothstep(a.v[i], lo.v[i], hi.v[i])), - arity: n, - } - } - Op::HsvAdjust { a, amount } => { - let hsv = rgb_to_hsv(g(*a).rgb()); - let m = g(*amount).rgb(); - hsv_to_rgb(Vec3A::new(hsv.x + m.x, hsv.y * m.y, hsv.z * m.z)).into() - } - Op::HeightToNormal { - xp, - xm, - yp, - ym, - scale, - } => height_to_normal( - g(*xp).x() - g(*xm).x(), - g(*yp).x() - g(*ym).x(), - g(*scale).x(), - ) - .into(), - } -} - -/// The chart coordinates `shift` footprint widths away from the shading -/// point. -/// -/// A zero shift returns `ctx.uv` untouched rather than adding `0 · width`, -/// which keeps every ordinary lookup bit-identical to what it was before -/// shifts existed (and to the JIT's inline texture path, which never sees a -/// shifted op). A shift over a zero footprint — no ray cone — lands back on -/// the shading point, so a derivative taken from shifted copies reads zero. -#[inline] -fn shifted_uv(ctx: &ShadeCtx, shift: [f32; 2]) -> (f32, f32) { - if shift == [0.0, 0.0] { - ctx.uv - } else { - ( - ctx.uv.0 + shift[0] * ctx.uv_width, - ctx.uv.1 + shift[1] * ctx.uv_width, - ) - } -} - -/// MaterialX's OSL `mx_heighttonormal_vector3`, given the height's change -/// across one footprint along `u` (`du`) and `v` (`dv`). -/// -/// The reference reads `dx = -Dx(in)`, `dy = Dy(in)`: screen-space -/// derivatives, i.e. the height's change across one pixel. The footprint is -/// this renderer's pixel (a ray cone's width in chart units), so the change -/// across it is the same quantity, with `Dx` running along `u`. Raster `y` -/// runs *down* while `v` runs up, so `Dy(in) = -dv` and both lateral -/// components come out as `-dh`: the normal of a surface raised by `in`. -/// That also makes the result resolution-dependent exactly as the -/// reference's is — a bump reads steeper the coarser the footprint. -fn height_to_normal(du: f32, dv: f32, scale: f32) -> Vec3A { - let (dx, dy) = (-du, -dv); - let dz = scale.max(1.0e-5) * (1.0 - dx * dx - dy * dy).max(1.0e-5).sqrt(); - Vec3A::new(dx, dy, dz).normalize_or(Vec3A::Z) * 0.5 + Vec3A::splat(0.5) -} - -/// MaterialX's `mx_rgbtohsv` (Foley & van Dam, via OSL), transcribed. -fn rgb_to_hsv(c: Vec3A) -> Vec3A { - let (r, g, b) = (c.x, c.y, c.z); - let min = r.min(g.min(b)); - let max = r.max(g.max(b)); - let delta = max - min; - let s = if max > 0.0 { delta / max } else { 0.0 }; - let h = if s <= 0.0 { - 0.0 - } else { - let h = if r >= max { - (g - b) / delta - } else if g >= max { - 2.0 + (b - r) / delta - } else { - 4.0 + (r - g) / delta - } * (1.0 / 6.0); - if h < 0.0 { h + 1.0 } else { h } - }; - Vec3A::new(h, s, max) -} - -/// MaterialX's `mx_hsvtorgb`, transcribed. The hue wraps, so a hue shift -/// past 1 comes round again. -fn hsv_to_rgb(hsv: Vec3A) -> Vec3A { - let (h, s, v) = (hsv.x, hsv.y, hsv.z); - if s < 0.0001 { - return Vec3A::splat(v); - } - let h = 6.0 * (h - h.floor()); - // `h` is in [0, 6) up to rounding; a non-finite hue lands in the last - // sextant rather than anywhere undefined. - let hi = h.trunc(); - let f = h - hi; - let p = v * (1.0 - s); - let q = v * (1.0 - s * f); - let t = v * (1.0 - s * (1.0 - f)); - match hi as i32 { - 0 => Vec3A::new(v, t, p), - 1 => Vec3A::new(q, v, p), - 2 => Vec3A::new(p, v, t), - 3 => Vec3A::new(p, q, v), - 4 => Vec3A::new(t, p, v), - _ => Vec3A::new(v, p, q), - } -} - -/// MaterialX `normalize`, over the value's own lanes. A zero-length input is -/// returned unchanged rather than divided into NaNs. A `float` broadcasts to a -/// `vector3`, as it always did here (MaterialX has no `float` variant). -fn normalize(a: Val) -> Val { - match a.arity { - 4 => { - let v = glam::Vec4::from_array(a.v); - let n = v.length(); - if n > 1e-20 { - let [x, y, z, w] = (v / n).to_array(); - Val::vec4(x, y, z, w) - } else { - a - } - } - arity => { - // Lanes past a `vector2`'s are whatever the op that made it left - // there, so they are zeroed before they can enter the length. - let v = if arity == 2 { - Vec3A::new(a.v[0], a.v[1], 0.0) - } else { - a.rgb() - }; - let n = v.length(); - if n > 1e-20 { - let r = v / n; - if arity == 2 { - Val::vec2(r.x, r.y) - } else { - r.into() - } - } else { - a - } - } - } -} - -/// MaterialX `dotproduct` over the operands' lanes. The `vector3` sum is -/// glam's, as it always was; the others extend it. (A `float` operand's -/// lanes all hold its value, so reading them broadcasts it.) -fn dot(a: Val, b: Val) -> Val { - let n = a.arity.max(b.arity); - let lanes = |v: Val| match n { - // A `vector2`'s third lane is not its own; see `normalize`. - 2 => Vec3A::new(v.v[0], v.v[1], 0.0), - _ => v.rgb(), - }; - let d = lanes(a).dot(lanes(b)); - Val::float(if n == 4 { d + a.v[3] * b.v[3] } else { d }) -} - -/// MaterialX `convert`. A `float` broadcasts. Widening a wider value fills -/// the new lanes the way the nodedefs do — zero, except that a `color4` / -/// `vector4` made from fewer lanes gets `1` in its last (an opaque alpha). -/// Narrowing keeps the leading lanes; to a `float`, the first. -fn convert(a: Val, arity: u8) -> Val { - let arity = arity.clamp(1, 4); - if a.arity == 1 { - return a.with_arity(arity); - } - if arity == 1 { - // Every lane of a `float` holds its value; see `Val::float`. - return Val::float(a.v[0]); - } - let mut v = a.v; - for (i, lane) in v.iter_mut().enumerate().skip(a.arity as usize) { - *lane = if i == 3 { 1.0 } else { 0.0 }; - } - Val { v, arity } -} - -/// MaterialX `combine2`: the lanes of `a`, then of `b`. Covers every -/// signature — `(float, float)` → `vector2`, `(color3, float)` → `color4`, -/// `(vector3, float)` and `(vector2, vector2)` → `vector4`. -fn combine2(a: Val, b: Val) -> Val { - let mut v = [0.0; 4]; - let (na, nb) = (a.arity as usize, b.arity as usize); - v[..na].copy_from_slice(&a.v[..na]); - let nb = nb.min(4 - na.min(4)); - v[na..na + nb].copy_from_slice(&b.v[..nb]); - Val { - v, - arity: (na + nb) as u8, - } -} - -/// MaterialX's `modulo`, OSL's `mod`: floored, so the result takes the -/// divisor's sign (`-0.2 mod 1` is `0.8`) where Rust's `%` truncates, and a -/// zero divisor returns the dividend, as OSL's does. -/// -/// OSL's `x − y·floor(x / y)` is kept wherever its quotient is finite, so -/// the rounding matches the reference (at `-1 mod -0.2` the quotient rounds -/// to 5 and the result to 0, a period away from the exact −0.19999999). Its -/// quotient overflows for some finite operands, though (`1 mod 1e-40` would -/// be `−inf`), and there the exact remainder `%` takes over, moved by one `y` -/// when its sign is the dividend's rather than the divisor's. -fn floored_mod(x: f32, y: f32) -> f32 { - if y == 0.0 { - return x; - } - let q = (x / y).floor(); - if q.is_finite() { - return x - y * q; - } - let r = x % y; - if r != 0.0 && (r < 0.0) != (y < 0.0) { - r + y - } else { - r - } -} - -/// OSL's `pow` (OIIO `safe_pow`), which MaterialX's `power` is: `x^0` is one, -/// `0^y` zero, a negative base takes only integer exponents (zero otherwise), -/// and the result is clamped finite. -fn safe_pow(x: f32, y: f32) -> f32 { - if y == 0.0 { - return 1.0; - } - if x == 0.0 { - return 0.0; - } - if x < 0.0 && y != y.floor() { - return 0.0; - } - x.powf(y).clamp(-f32::MAX, f32::MAX) -} - -/// OSL's `smoothstep(low, high, x)`, which MaterialX's is: zero below `low`, -/// one from `high` up, the Hermite ramp between. With `low >= high` the first -/// two tests decide every `x`, so no division by the empty interval happens. -fn smoothstep(x: f32, low: f32, high: f32) -> f32 { - if x < low { - 0.0 - } else if x >= high { - 1.0 - } else { - let t = (x - low) / (high - low); - t * t * (3.0 - 2.0 * t) - } -} - -/// MaterialX `normalmap`: decode `[0,1]`-encoded tangent-space vector, scale -/// its lateral components, and rotate it into world space. -/// -/// Falls back to the geometric normal when there is no tangent — the host -/// says when that happens (in crust, only baked single-placement geometry -/// carries one; see crust-core's `UvMap::tangents`). Returning the -/// geometric normal is the right degradation: a normal map's *mean* is the -/// surface normal, so the flat surface is the map's own zero. -fn normal_map(encoded: Val, scale: Val, ctx: &ShadeCtx) -> Vec3A { - let v = encoded.rgb() * 2.0 - Vec3A::ONE; - // `scale` is a `float` or, per axis, a `vector2`. - let (sx, sy) = if scale.arity >= 2 { - (scale.v[0], scale.v[1]) - } else { - (scale.x(), scale.x()) - }; - let finite = |s: f32| if s.is_finite() { s } else { 1.0 }; - let local = Vec3A::new(v.x * finite(sx), v.y * finite(sy), v.z.max(1e-4)); - perturb_normal(local, ctx.normal, ctx.tangent) -} - -/// Rotates a **decoded** tangent-space normal (`z` along `normal`) into world -/// space, against `tangent` re-orthogonalised to `normal`. -/// -/// The half of `normal_map` that knows nothing about MaterialX's `[0,1]` -/// encoding, public so a host with its own decode — UsdPreviewSurface's -/// `normal` input arrives already in `[-1,1]`, its UsdUVTexture's -/// `scale`/`bias` having done the decode — rotates it identically. Returns -/// `normal` unchanged when `tangent` is zero (no chart frame) or parallel to -/// it, for the reason `normal_map` gives. -pub fn perturb_normal(local: Vec3A, normal: Vec3A, tangent: Vec3A) -> Vec3A { - if tangent.length_squared() < 1e-20 { - return normal; - } - let n = normal; - // Re-orthogonalise: the stored tangent is the triangle's, while `n` may - // already carry interpolated shading curvature, so the two need not be - // perpendicular. - let t = (tangent - n * n.dot(tangent)).normalize_or_zero(); - if t.length_squared() < 1e-20 { - return n; - } - let b = n.cross(t); - let world = t * local.x + b * local.y + n * local.z; - if world.length_squared() > 1e-20 { - world.normalize() - } else { - n - } -} - -/// Gulbrandsen's "Artist Friendly Metallic Fresnel": normal-incidence -/// reflectivity and grazing edge tint → complex IOR. -/// -/// Implemented rather than short-circuited (the conductor lobe wants a -/// reflectivity back, which is what went in) because a graph may author `ior` -/// and `extinction` directly, and the round trip through -/// [`reflectivity_from_ior`] then handles both authorings with one path. -fn artistic_ior(reflectivity: Vec3A, edge: Vec3A) -> (Vec3A, Vec3A) { - let r = reflectivity.clamp(Vec3A::ZERO, Vec3A::splat(0.99)); - let rs = Vec3A::new(r.x.sqrt(), r.y.sqrt(), r.z.sqrt()); - let n_min = (Vec3A::ONE - r) / (Vec3A::ONE + r); - let n_max = (Vec3A::ONE + rs) / (Vec3A::ONE - rs).max(Vec3A::splat(1e-6)); - // OSL's `mix(n_max, n_min, edge)`, as `x·(1 − t) + y·t`: an edge colour - // of white — the default — selects `n_min` exactly, where `x + (y − x)·t` - // would leave `n_max`'s rounding in it (n_max is ~70 for a bright metal). - // Clamped, unlike the reference: an edge tint outside [0, 1] extrapolates - // the IOR to nonsense, down to negative values. - let e = edge.clamp(Vec3A::ZERO, Vec3A::ONE); - let n = n_max * (Vec3A::ONE - e) + n_min * e; - let np1 = n + Vec3A::ONE; - let nm1 = n - Vec3A::ONE; - let k2 = - ((np1 * np1 * r - nm1 * nm1) / (Vec3A::ONE - r).max(Vec3A::splat(1e-6))).max(Vec3A::ZERO); - (n, Vec3A::new(k2.x.sqrt(), k2.y.sqrt(), k2.z.sqrt())) -} - -/// Normal-incidence reflectivity of a conductor with complex IOR `n + ik`. -/// -/// The exact inverse of `artistic_ior` (the node's implementation above), which -/// is what lets a conductor lobe -/// be reduced to the one colour OpenPBR's metal lobe takes, whichever way the -/// graph authored it. -pub fn reflectivity_from_ior(n: Vec3A, k: Vec3A) -> Vec3A { - let num = (n - Vec3A::ONE) * (n - Vec3A::ONE) + k * k; - let den = ((n + Vec3A::ONE) * (n + Vec3A::ONE) + k * k).max(Vec3A::splat(1e-6)); - (num / den).clamp(Vec3A::ZERO, Vec3A::ONE) -} - -// --------------------------------------------------------------------------- -// Compilation -// --------------------------------------------------------------------------- - -/// `luminance`'s default `lumacoeffs`: ACEScg's (AP1) weights. -const AP1_LUMA_COEFFS: Val = Val::vec3(0.2722287, 0.6740818, 0.0536895); - -/// Turns named `.mtlx` nodes into a topologically ordered [`Program`]. -pub struct Compiler<'a> { - pub doc: &'a Doc, - pub program: Program, - /// Slot already emitted for a `(graph, node, output, shift)` key, so a - /// node feeding five others is evaluated once. The output name matters: - /// `artistic_ior` emits a different slot for `ior` than for `extinction`. - /// So does the shift: a node under `heighttonormal` is compiled once per - /// offset it is differentiated at. - memo: std::collections::HashMap<(String, String, String, [u32; 2]), u32>, - /// The offset, in footprint widths, every texture lookup and `texcoord` - /// compiled now is displaced by — zero except while `heighttonormal` - /// compiles the shifted copies of its input. - uv_shift: [f32; 2], - /// What the loader answered per `(file, colorspace)` — the space as - /// handed to it — so the shifted copies of an `image`, and two images of - /// one file in one effective space, share the one sampler rather than - /// asking the host again. - images: std::collections::HashMap<(String, Option), Option>, - /// Nodes currently being compiled, so a cyclic document — which a - /// hand-edited `.mtlx` can be — terminates as a constant rather than - /// recursing until the stack runs out. - active: Vec, - /// Resolves an `image` node's `file` input to a sampler, and converts - /// authored colours into the working space. - host: crate::Host<'a>, - /// Node categories met that this compiler has no operator for, for one - /// summary warning instead of one per occurrence. - pub unsupported: std::collections::BTreeSet, -} - -impl<'a> Compiler<'a> { - pub fn new(doc: &'a Doc, host: &crate::Host<'a>) -> Compiler<'a> { - Compiler { - doc, - program: Program::default(), - memo: std::collections::HashMap::new(), - uv_shift: [0.0, 0.0], - images: std::collections::HashMap::new(), - active: Vec::new(), - host: *host, - unsupported: Default::default(), - } - } - - pub fn emit(&mut self, op: Op) -> u32 { - self.program.ops.push(op); - (self.program.ops.len() - 1) as u32 - } - - pub fn constant(&mut self, v: Val) -> u32 { - self.emit(Op::Const(v)) - } - - /// `luminance` at the nodedef's default `lumacoeffs`, ACEScg's (AP1) — - /// what the stdlib nodegraphs' unauthored `luminance` / `saturate` read. - pub fn luminance(&mut self, a: u32) -> u32 { - let coeffs = self.constant(AP1_LUMA_COEFFS); - self.emit(Op::Luminance { a, coeffs }) - } - - /// The value of `slot` when it is a compile-time constant — a literal, or - /// a pure operator over constants — and `None` when it depends on the - /// shading point or a texture. - /// - /// What the closure builders prune on: a branch whose weight folds to a - /// literal zero, or a mix whose factor folds to exactly 0 or 1, is left - /// out of the tree altogether rather than evaluated at every vertex to be - /// told it contributes nothing. - pub fn fold(&self, slot: u32) -> Option { - let ops = &self.program.ops; - let op = ops.get(slot as usize)?; - if let Op::Const(v) = op { - return Some(*v); - } - if !op.is_pure() { - return None; - } - let mut operands = Vec::new(); - op.clone().for_each_operand(|o| operands.push(*o)); - let mut slots = vec![Val::ZERO; slot as usize]; - for o in operands { - slots[o as usize] = self.fold(o)?; - } - let ctx = ShadeCtx { - uv: (0.0, 0.0), - normal: Vec3A::Z, - tangent: Vec3A::X, - view: -Vec3A::Z, - position: Vec3A::ZERO, - uv_width: 0.0, - }; - Some(apply(op, &slots, &ctx)) - } - - /// Compiles the value feeding `input` of `node`, or `default` when the - /// input is unauthored. - pub fn input_or(&mut self, node: &Node, name: &str, default: Val) -> u32 { - match node.input(name) { - Some(i) => self.compile_input(node, i), - // The nodedef's default is typed: an unauthored `in1` of an - // `add_color3` is a colour3 zero, not a `float` one, and an op over - // nothing but defaults must come out at the node's width. A - // `float` default's lanes already hold its value, so widening it - // changes no lane — only what a width-reading consumer - // (`luminance`, `normalize`, `dotproduct`, `combine2`) sees. Not - // `convert`, whose input width is what decides the alpha. - None if default.arity == 1 && node.category != "convert" => { - self.constant(default.broadcast_to(arity_of(&node.type_name))) - } - None => self.constant(default), - } - } - - /// Like [`Compiler::input_or`] but reports whether the input was authored, - /// which the BSDF reducer needs to tell "no normal map" from "a normal map - /// that happens to be flat". - pub fn optional_input(&mut self, node: &Node, name: &str) -> Option { - let i = node.input(name)?; - Some(self.compile_input(node, i)) - } - - /// The width an authored input carries: its producer's declared type - /// when it is connected, since the parser reads an input with no `type` - /// attribute as a `float` whatever feeds it, and the input's own type for - /// a literal. A multioutput producer's outputs are not typed in the - /// document, so there the input's declaration is all there is. - fn input_arity(&self, node: &Node, input: &Input) -> u8 { - let scope = node.graph.clone().unwrap_or_default(); - let producer = match &input.source { - Source::Node { name, .. } => self.doc.find(&scope, name), - Source::Graph { graph, output } => self.doc.graph_output(graph, output).map(|c| c.node), - Source::Value(_) => None, - }; - match producer { - Some(p) if p.type_name != "multioutput" => arity_of(&p.type_name), - _ => arity_of(&input.type_name), - } - } - - fn compile_input(&mut self, node: &Node, input: &Input) -> u32 { - let scope = node.graph.clone().unwrap_or_default(); - match &input.source { - Source::Value(v) => { - // A literal "0.5" on a color3 input broadcasts, which the - // parser already arranged; a colour is then taken into the - // working space. - let v = self.managed(node, input, *v); - self.constant(v) - } - Source::Node { name, output } => self.compile_named(&scope, name, output.as_deref()), - Source::Graph { graph, output } => match self.doc.graph_output(graph, output) { - Some(conn) => { - // The graph's `` may itself select one output of a - // multioutput node; carry it through, or `extinction` - // silently compiles as `ior`. - let (g, nm) = ( - conn.node.graph.clone().unwrap_or_default(), - conn.node.name.clone(), - ); - let sel = conn.output.map(str::to_string); - self.compile_named(&g, &nm, sel.as_deref()) - } - None => self.constant(Val::ZERO), - }, - } - } - - /// Compiles the node called `name`, returning its slot. - pub fn compile_named(&mut self, scope: &str, name: &str, output: Option<&str>) -> u32 { - let key = ( - scope.to_string(), - name.to_string(), - output.unwrap_or("").to_string(), - self.uv_shift.map(f32::to_bits), - ); - if let Some(&slot) = self.memo.get(&key) { - return slot; - } - if self.active.iter().any(|a| a == name) { - // A cycle. Break it with a constant — the alternative is a stack - // overflow at import on a malformed document. - return self.constant(Val::ZERO); - } - let Some(node) = self.doc.find(scope, name) else { - return self.constant(Val::ZERO); - }; - let node = node.clone(); - self.active.push(name.to_string()); - let slot = self.compile_node(&node, output); - self.active.pop(); - self.memo.insert(key, slot); - slot - } - - fn compile_node(&mut self, node: &Node, output: Option<&str>) -> u32 { - let arity = arity_of(&node.type_name); - let bin = |c: &mut Self, op: BinOp, d1: Val, d2: Val| { - let a = c.input_or(node, "in1", d1); - let b = c.input_or(node, "in2", d2); - c.emit(Op::Binary { op, a, b }) - }; - let un = |c: &mut Self, op: UnOp| { - let a = c.input_or(node, "in", Val::ZERO); - c.emit(Op::Unary { op, a }) - }; - match node.category.as_str() { - "constant" => self.input_or(node, "value", Val::ZERO), - "image" | "tiledimage" => self.compile_image(node, arity), - "texcoord" => self.emit(Op::TexCoord { - shift: self.uv_shift, - }), - "normal" => self.emit(Op::Normal), - "viewdirection" => self.emit(Op::ViewDirection), - "position" => self.emit(Op::Position), - "add" => bin(self, BinOp::Add, Val::ZERO, Val::ZERO), - "subtract" => bin(self, BinOp::Sub, Val::ZERO, Val::ZERO), - // The unauthored defaults are the nodedefs': `in1` is zero and - // `in2` one for every operator whose identity is one. - "multiply" => bin(self, BinOp::Mul, Val::ZERO, Val::ONE), - "divide" => bin(self, BinOp::Div, Val::ZERO, Val::ONE), - "power" => bin(self, BinOp::Pow, Val::ZERO, Val::ONE), - "min" => bin(self, BinOp::Min, Val::ZERO, Val::ZERO), - "max" => bin(self, BinOp::Max, Val::ZERO, Val::ZERO), - "modulo" => bin(self, BinOp::Modulo, Val::ZERO, Val::ONE), - "absval" => un(self, UnOp::Abs), - "ln" => { - let a = self.input_or(node, "in", Val::ONE); - self.emit(Op::Unary { op: UnOp::Ln, a }) - } - "exp" => un(self, UnOp::Exp), - "sin" => un(self, UnOp::Sin), - "cos" => un(self, UnOp::Cos), - "asin" => un(self, UnOp::Asin), - "acos" => un(self, UnOp::Acos), - "sqrt" => un(self, UnOp::Sqrt), - "sign" => un(self, UnOp::Sign), - "floor" => un(self, UnOp::Floor), - "ceil" => un(self, UnOp::Ceil), - "normalize" => un(self, UnOp::Normalize), - "luminance" => { - let a = self.input_or(node, "in", Val::ZERO); - // The nodedef's default: ACEScg's (AP1) coefficients. - let coeffs = self.input_or(node, "lumacoeffs", AP1_LUMA_COEFFS); - self.emit(Op::Luminance { a, coeffs }) - } - "dotproduct" => { - let a = self.input_or(node, "in1", Val::ZERO); - let b = self.input_or(node, "in2", Val::ZERO); - self.emit(Op::DotProduct { a, b }) - } - "mix" => { - let fg = self.input_or(node, "fg", Val::ZERO); - let bg = self.input_or(node, "bg", Val::ZERO); - let m = self.input_or(node, "mix", Val::ZERO); - self.emit(Op::Mix { fg, bg, m }) - } - "clamp" => { - let a = self.input_or(node, "in", Val::ZERO); - let low = self.input_or(node, "low", Val::ZERO); - let high = self.input_or(node, "high", Val::ONE); - self.emit(Op::Clamp { a, low, high }) - } - "contrast" => { - let a = self.input_or(node, "in", Val::ZERO); - let amount = self.input_or(node, "amount", Val::ONE); - let pivot = self.input_or(node, "pivot", Val::float(0.5)); - self.emit(Op::Contrast { a, amount, pivot }) - } - "remap" => { - let a = self.input_or(node, "in", Val::ZERO); - let in_low = self.input_or(node, "inlow", Val::ZERO); - let in_high = self.input_or(node, "inhigh", Val::ONE); - let out_low = self.input_or(node, "outlow", Val::ZERO); - let out_high = self.input_or(node, "outhigh", Val::ONE); - self.emit(Op::Remap { - a, - in_low, - in_high, - out_low, - out_high, - }) - } - "invert" => { - let a = self.input_or(node, "in", Val::ZERO); - let amount = self.input_or(node, "amount", Val::ONE); - self.emit(Op::Invert { a, amount }) - } - "smoothstep" => { - let a = self.input_or(node, "in", Val::ZERO); - let low = self.input_or(node, "low", Val::ZERO); - let high = self.input_or(node, "high", Val::ONE); - self.emit(Op::Smoothstep { a, low, high }) - } - "convert" => { - let a = self.input_or(node, "in", Val::ZERO); - self.emit(Op::Convert { a, arity }) - } - "extract" => { - let a = self.input_or(node, "in", Val::ZERO); - // `index` is an integer literal, so it is read off the raw - // text rather than through the float lanes. - let index = node - .input("index") - .and_then(|i| i.text.as_deref()) - .and_then(|t| t.trim().parse::().ok()) - .unwrap_or(0); - self.emit(Op::Extract { a, index }) - } - "combine2" => { - // The signature fixes each operand's width — `(float, float)` - // → `vector2`, `(color3, float)` → `color4`, and for a - // `vector4` `(vector3, float)` or `(vector2, vector2)`, told - // apart by either input's width (see `Compiler::input_arity`). - // Each operand is converted to its width first, so the - // concatenation is exact whatever produced it (an unauthored - // `in1` is a zero, which must still fill three lanes of a - // `color4`). - let declared = |n: &str| node.input(n).map(|i| self.input_arity(node, i)); - let (na, nb) = match arity { - 4 if declared("in1") == Some(2) || declared("in2") == Some(2) => (2, 2), - 4 => (3, 1), - _ => (1, 1), - }; - let a = self.input_or(node, "in1", Val::ZERO); - let a = self.emit(Op::Convert { a, arity: na }); - let b = self.input_or(node, "in2", Val::ZERO); - let b = self.emit(Op::Convert { a: b, arity: nb }); - self.emit(Op::Combine2 { a, b }) - } - "combine3" => { - let a = self.input_or(node, "in1", Val::ZERO); - let b = self.input_or(node, "in2", Val::ZERO); - let c = self.input_or(node, "in3", Val::ZERO); - self.emit(Op::Combine3 { a, b, c }) - } - "normalmap" => { - let a = self.input_or(node, "in", Val::vec3(0.5, 0.5, 1.0)); - let scale = self.input_or(node, "scale", Val::ONE); - self.emit(Op::NormalMap { a, scale }) - } - "artistic_ior" => { - let reflectivity = - self.input_or(node, "reflectivity", Val::vec3(0.944, 0.776, 0.373)); - let edge = self.input_or(node, "edge_color", Val::vec3(0.998, 0.981, 0.751)); - self.emit(Op::ArtisticIor { - reflectivity, - edge, - extinction: output == Some("extinction"), - }) - } - "chiang_hair_roughness" => self.compile_chiang_hair_roughness(node, output), - "chiang_hair_absorption_from_color" => { - self.compile_chiang_hair_absorption_from_color(node) - } - "deon_hair_absorption_from_melanin" => { - self.compile_deon_hair_absorption_from_melanin(node) - } - "colorcorrect" => self.compile_colorcorrect(node), - "heighttonormal" => { - let scale = self.input_or(node, "scale", Val::ONE); - // Half a footprint either side, so the difference spans one. - let xp = self.shifted_input(node, "in", [0.5, 0.0]); - let xm = self.shifted_input(node, "in", [-0.5, 0.0]); - let yp = self.shifted_input(node, "in", [0.0, 0.5]); - let ym = self.shifted_input(node, "in", [0.0, -0.5]); - self.emit(Op::HeightToNormal { - xp, - xm, - yp, - ym, - scale, - }) - } - other => { - self.unsupported.insert(other.to_string()); - // Degrade this input to mid-grey rather than to black: an - // unsupported *pattern* node is usually a colour correction, - // and zero would turn whatever it feeds into a hole. - self.constant(Val::float(0.5)) - } - } - } - - /// `input` of `node` compiled with every lookup under it displaced by a - /// further `shift` footprint widths. - fn shifted_input(&mut self, node: &Node, input: &str, shift: [f32; 2]) -> u32 { - let outer = self.uv_shift; - self.uv_shift = [outer[0] + shift[0], outer[1] + shift[1]]; - let slot = self.input_or(node, input, Val::ZERO); - self.uv_shift = outer; - slot - } - - /// MaterialX `colorcorrect` (`color3`), lowered to the stdlib's own - /// `NG_colorcorrect_color3` chain: `hsvadjust` (hue) → `saturate` → - /// `range` (gamma) → lift → gain → `contrast` → exposure. - /// - /// Expanded here rather than kept as one op so the stages reuse - /// operators the JIT already inlines. A stage whose parameter folds to - /// its identity (hue 0, saturation 1, …) is left out: the reference's - /// arithmetic at those values is the identity up to rounding at most, and - /// the playground's graphs author one or two of the eight inputs. - fn compile_colorcorrect(&mut self, node: &Node) -> u32 { - if node.type_name != "color3" { - // `color4` routes alpha around the chain, and there is no - // `combine4` to put it back with; say so rather than correct the - // alpha too. - self.unsupported - .insert(format!("colorcorrect ({})", node.type_name)); - return self.constant(Val::float(0.5)); - } - let input = self.input_or(node, "in", Val::vec3(1.0, 1.0, 1.0)); - // Promote a `float` input to the colour MaterialX makes of it — its - // first lane, three times — before any stage runs. A one-lane texture - // carries its file's other channels in the lanes above the first, and - // the stages below work lane by lane, so without this they would - // correct those hidden channels too and a later `convert` would - // surface them as colour. `zip` broadcasts a one-lane operand from - // lane 0 and multiplying by one is exact, so a `color3` input passes - // through bit for bit. - let ones = self.constant(Val::vec3(1.0, 1.0, 1.0)); - let mut c = self.emit(Op::Binary { - op: BinOp::Mul, - a: input, - b: ones, - }); - // Each parameter's slot, unless it folds to the stage's identity. - let param = |cc: &mut Self, name: &str, default: f32| -> Option { - let s = cc.input_or(node, name, Val::float(default)); - (cc.fold(s).map(|v| v.x()) != Some(default)).then_some(s) - }; - if let Some(hue) = param(self, "hue", 0.0) { - let one = self.constant(Val::ONE); - let amount = self.emit(Op::Combine3 { - a: hue, - b: one, - c: one, - }); - c = self.emit(Op::HsvAdjust { a: c, amount }); - } - if let Some(sat) = param(self, "saturation", 1.0) { - // `saturate`: mix from the luminance grey toward the colour. - let grey = self.luminance(c); - c = self.emit(Op::Mix { - fg: c, - bg: grey, - m: sat, - }); - } - if let Some(gamma) = param(self, "gamma", 1.0) { - // `range` over [0, 1] → [0, 1] unclamped: both remaps are the - // identity, leaving `sign(x)·|x|^(1/gamma)`. - let one = self.constant(Val::ONE); - let recip = self.emit(Op::Binary { - op: BinOp::Div, - a: one, - b: gamma, - }); - let abs = self.emit(Op::Unary { - op: UnOp::Abs, - a: c, - }); - let pow = self.emit(Op::Binary { - op: BinOp::Pow, - a: abs, - b: recip, - }); - let sign = self.emit(Op::Unary { - op: UnOp::Sign, - a: c, - }); - c = self.emit(Op::Binary { - op: BinOp::Mul, - a: pow, - b: sign, - }); - } - if let Some(lift) = param(self, "lift", 0.0) { - // `c·(1 − lift) + lift`: raises black to `lift`, keeps white. - let one = self.constant(Val::ONE); - let keep = self.emit(Op::Binary { - op: BinOp::Sub, - a: one, - b: lift, - }); - let scaled = self.emit(Op::Binary { - op: BinOp::Mul, - a: c, - b: keep, - }); - c = self.emit(Op::Binary { - op: BinOp::Add, - a: scaled, - b: lift, - }); - } - if let Some(gain) = param(self, "gain", 1.0) { - c = self.emit(Op::Binary { - op: BinOp::Mul, - a: c, - b: gain, - }); - } - if let Some(amount) = param(self, "contrast", 1.0) { - let pivot = self.input_or(node, "contrastpivot", Val::float(0.5)); - c = self.emit(Op::Contrast { - a: c, - amount, - pivot, - }); - } - if let Some(exposure) = param(self, "exposure", 0.0) { - let two = self.constant(Val::float(2.0)); - let k = self.emit(Op::Binary { - op: BinOp::Pow, - a: two, - b: exposure, - }); - c = self.emit(Op::Binary { - op: BinOp::Mul, - a: c, - b: k, - }); - } - c - } - - fn compile_image(&mut self, node: &Node, arity: u8) -> u32 { - let file = node.input("file"); - // The texels' colour space: the `file` input's effective one, for a - // colour image only. A `float` or `vector3` image is data — a mask, - // a height, a normal map — whatever the document's default says. - let space = file - .filter(|_| is_color_type(&node.type_name)) - .and_then(|f| self.doc.colorspace_of(node, f)) - .map(str::to_string); - let tex = file.and_then(|i| i.text.clone()).and_then(|f| { - let loader = self.host.load_texture; - self.images - .entry((f, space)) - .or_insert_with_key(|(f, s)| loader(f, s.as_deref())) - .clone() - }); - // The authored `default` is a literal like any other: a colour one is - // converted from its own effective space. - let fallback = self.literal_of(node, "default").unwrap_or(Val::float(0.5)); - // `tiledimage` scales and offsets the chart before the lookup; - // `image` samples it as authored. - let (scale, offset) = if node.category == "tiledimage" { - let s = self.static_input(node, "uvtiling", Val::vec2(1.0, 1.0)); - let o = self.static_input(node, "uvoffset", Val::vec2(0.0, 0.0)); - let s = if s.arity == 1 { - [s.x(), s.x()] - } else { - [s.v[0], s.v[1]] - }; - let o = if o.arity == 1 { - [o.x(), o.x()] - } else { - [o.v[0], o.v[1]] - }; - (s, o) - } else { - ([1.0, 1.0], [0.0, 0.0]) - }; - let coord = self.image_coord(node); - self.emit(Op::Texture { - tex, - fallback, - scale, - offset, - arity, - shift: self.uv_shift, - coord, - }) - } - - /// An image's authored `texcoord`, compiled, or `None` for the shading - /// point's own chart. - /// - /// `None` covers the unconnected input and a connection to the default - /// chart itself (`texcoord` index 0, or `geompropvalue` of `st`), which is - /// how nearly every document spells it; those keep the op on the JIT's - /// inline texture path. A connection whose subgraph meets an operator - /// this compiler lacks (`place2d`, a second UV set) also takes `None`: - /// the chart is a better stand-in for an unknown coordinate than the - /// constant the unknown node would compile to, and that node is already - /// reported. - fn image_coord(&mut self, node: &Node) -> Option { - let input = node.input("texcoord")?; - if let Source::Node { name, .. } = &input.source { - let scope = node.graph.clone().unwrap_or_default(); - if self.doc.find(&scope, name).is_some_and(is_default_chart) { - return None; - } - } - // A literal compiles to a constant: one fixed texel, as authored. - let reported = self.unsupported.len(); - let slot = self.compile_input(node, input); - (self.unsupported.len() == reported).then_some(slot) - } -} - -impl Compiler<'_> { - /// An authored literal `v` of `input` as the program holds it: a - /// `color3` / `color4` with an effective colour space converted into the - /// working space by the host, anything else as authored. - fn managed(&self, node: &Node, input: &Input, v: Val) -> Val { - if !is_color_type(&input.type_name) { - return v; - } - match self.doc.colorspace_of(node, input) { - Some(space) => convert_color(v, &input.type_name, |rgb| { - (self.host.convert_color)(space, rgb) - }), - None => v, - } - } - - /// `node`'s input `name` when it is an authored literal (not a - /// connection), with the value the program holds for it — converted, for - /// a colour. What literal-driven pruning reads, so that it tests exactly - /// the value every shading point would have computed. - pub(crate) fn literal_of(&self, node: &Node, name: &str) -> Option { - let input = node.input(name)?; - match input.source { - Source::Value(v) => Some(self.managed(node, input, v)), - _ => None, - } - } - - /// An input the lookup needs as a compile-time constant: the literal, or - /// a connection that folds to one (`convert(8.0)` is how documents feed a - /// `tiledimage`'s `uvtiling`). One that varies over the surface cannot be - /// baked into the texture op, so it takes `default` and is reported — - /// never silently. - fn static_input(&mut self, node: &Node, name: &str, default: Val) -> Val { - if let Some(v) = self.literal_of(node, name) { - return v; - } - if node.input(name).is_none() { - return default; - } - let slot = self.input_or(node, name, default); - self.fold(slot).unwrap_or_else(|| { - self.unsupported - .insert(format!("{} (varying {name})", node.category)); - default - }) - } -} - -/// Whether `node` is the shading point's default chart: `texcoord` with -/// index 0, or `geompropvalue` reading `st` — what `ShadeCtx::uv` holds. -fn is_default_chart(node: &Node) -> bool { - let text = |name: &str| { - node.input(name) - .and_then(|i| i.text.as_deref()) - .map(str::trim) - }; - match node.category.as_str() { - "texcoord" => text("index").is_none_or(|i| i == "0"), - "geompropvalue" => text("geomprop") == Some("st"), - _ => false, - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::Host; - - fn no_textures(_: &str, _: Option<&str>) -> Option { - None - } - - fn run(doc_text: &str, node: &str) -> Val { - let doc = Doc::parse(doc_text).unwrap(); - let loader = no_textures; - let mut c = Compiler::new(&doc, &Host::new(&loader)); - let slot = c.compile_named("", node, None); - let mut slots = Vec::new(); - c.program.eval( - &ShadeCtx { - uv: (0.25, 0.75), - normal: Vec3A::Z, - tangent: Vec3A::X, - view: -Vec3A::Z, - position: Vec3A::ZERO, - uv_width: 0.0, - }, - &mut slots, - ); - slots[slot as usize] - } - - #[test] - fn mix_blends_bg_toward_fg() { - let v = run( - r#" - - - - - - - - "#, - "m", - ); - assert!((v.x() - 0.25).abs() < 1e-6, "got {}", v.x()); - } - - #[test] - fn artistic_ior_round_trips_to_its_reflectivity() { - // The conductor lobe reduces (n, k) back to a reflectivity colour, so - // the two conversions must be inverses or every metal shifts hue. - for r in [0.05f32, 0.4, 0.94] { - let (n, k) = artistic_ior(Vec3A::splat(r), Vec3A::ONE); - let back = reflectivity_from_ior(n, k); - assert!((back.x - r).abs() < 1e-4, "r={r} -> {}", back.x); - } - } - - #[test] - fn a_graph_output_keeps_the_output_it_selected() { - // A ``'s `` may select one output of a multioutput - // node. Dropping that selection is silent: the reference falls back to - // the node's first output, so a graph publishing `artistic_ior`'s - // extinction hands its consumer the ior instead. - let doc = r#" - - - - - - - - - - - - - - - - - "#; - let (n, k) = artistic_ior(Vec3A::splat(0.5), Vec3A::ONE); - let got_n = run(doc, "take_n"); - let got_k = run(doc, "take_k"); - assert!((got_n.x() - n.x).abs() < 1e-5, "ior: got {}", got_n.x()); - assert!( - (got_k.x() - k.x).abs() < 1e-5, - "extinction: got {}", - got_k.x() - ); - // The two outputs are what the test is about; equal values would make - // the assertions above pass for the wrong reason. - assert!((n.x - k.x).abs() > 1e-3); - } - - #[test] - fn a_graph_output_without_a_selection_takes_the_first_output() { - // The single-output majority authors no `output` attribute, and must - // keep resolving as it did. - let v = run( - r#" - - - - - - - - - - - - "#, - "take", - ); - let (n, _) = artistic_ior(Vec3A::splat(0.5), Vec3A::ONE); - assert!((v.x() - n.x).abs() < 1e-5, "got {}", v.x()); - } - - #[test] - fn a_cycle_terminates_instead_of_overflowing_the_stack() { - let v = run( - r#" - - - - - - - "#, - "a", - ); - assert!(v.x().is_finite()); - } - - /// `colorcorrect` of a constant `in`, with `params` authored as floats. - fn colorcorrect(input: [f32; 3], params: &[(&str, f32)]) -> Vec3A { - let inputs: String = params - .iter() - .map(|(n, v)| format!(r#""#)) - .collect(); - let doc = format!( - r#" - - - {inputs} - - "#, - input[0], input[1], input[2] - ); - let v = run(&doc, "cc"); - assert_eq!(v.arity, 3); - v.rgb() - } - - fn close(got: Vec3A, want: Vec3A) { - assert!( - (got - want).abs().max_element() < 1e-5, - "got {got}, want {want}" - ); - } - - #[test] - fn colorcorrect_defaults_are_the_identity() { - // Bitwise: every stage folds to its identity and is left out. - let c = [0.1, 0.7, 2.5]; - assert_eq!(colorcorrect(c, &[]), Vec3A::from(c)); - assert_eq!( - colorcorrect(c, &[("hue", 0.0), ("gain", 1.0), ("exposure", 0.0)]), - Vec3A::from(c) - ); - } - - #[test] - fn colorcorrect_applies_each_stage_as_the_stdlib_graph_does() { - let c = [0.125, 0.5, 1.0]; - close(colorcorrect(c, &[("gain", 4.0)]), Vec3A::new(0.5, 2.0, 4.0)); - // `range` with gamma: x^(1/gamma), sign-preserving, unclamped. - close( - colorcorrect([0.125, 8.0, -0.125], &[("gamma", 3.0)]), - Vec3A::new(0.5, 2.0, -0.5), - ); - // Lift raises black to `lift` and leaves white alone. - close( - colorcorrect([0.0, 0.5, 1.0], &[("lift", 0.5)]), - Vec3A::new(0.5, 0.75, 1.0), - ); - close( - colorcorrect( - [0.25, 0.5, 1.0], - &[("contrast", 2.0), ("contrastpivot", 0.5)], - ), - Vec3A::new(0.0, 0.5, 1.5), - ); - close( - colorcorrect(c, &[("exposure", -1.0)]), - Vec3A::new(0.0625, 0.25, 0.5), - ); - // Saturation 0 is the luminance grey (MaterialX's ACEScg default - // `lumacoeffs`); 2 pushes away from it. - let l = 0.2722287 * 0.125 + 0.6740818 * 0.5 + 0.0536895; - close(colorcorrect(c, &[("saturation", 0.0)]), Vec3A::splat(l)); - close( - colorcorrect(c, &[("saturation", 2.0)]), - Vec3A::from(c) * 2.0 - Vec3A::splat(l), - ); - // Hue rotates in turns and wraps: a third of a turn takes red to - // green, and 1.5 turns is the same as a half. - close( - colorcorrect([1.0, 0.0, 0.0], &[("hue", 1.0 / 3.0)]), - Vec3A::new(0.0, 1.0, 0.0), - ); - close( - colorcorrect([1.0, 0.0, 0.0], &[("hue", 1.5)]), - Vec3A::new(0.0, 1.0, 1.0), - ); - } - - #[test] - fn colorcorrect_stages_run_in_the_stdlib_order() { - // Gamma before lift before gain before contrast before exposure: any - // two swapped gives a different answer on this input. - let got = colorcorrect( - [0.25; 3], - &[ - ("gamma", 2.0), - ("lift", 0.5), - ("gain", 2.0), - ("contrast", 0.5), - ("exposure", 1.0), - ], - ); - let x = 0.25f32.sqrt(); // gamma → 0.5 - let x = x * (1.0 - 0.5) + 0.5; // lift → 0.75 - let x = x * 2.0; // gain → 1.5 - let x = (x - 0.5) * 0.5 + 0.5; // contrast → 1.0 - let x = x * 2.0; // exposure → 2.0 - close(got, Vec3A::splat(x)); - } - - #[test] - fn hsv_round_trips() { - for c in [ - Vec3A::new(0.9, 0.2, 0.1), - Vec3A::new(0.1, 0.8, 0.3), - Vec3A::new(0.2, 0.3, 0.95), - Vec3A::new(0.7, 0.1, 0.6), - Vec3A::splat(0.4), - Vec3A::new(3.0, 1.0, 0.5), - ] { - close(hsv_to_rgb(rgb_to_hsv(c)), c); - } - } - - /// A height of `slope_u · u + slope_v · v`, sampled through a real - /// `image` node so the shifted lookups are what is being differentiated. - struct Ramp { - slope_u: f32, - slope_v: f32, - } - impl crate::Texture for Ramp { - fn eval(&self, u: f32, v: f32, _: f32) -> [f32; 4] { - [self.slope_u * u + self.slope_v * v; 4] - } - } - - const HEIGHT_DOC: &str = r#" - - - - - - - - - - - "#; - - fn height_to_normal_at(slope_u: f32, slope_v: f32, uv_width: f32, node: &str) -> Vec3A { - let doc = Doc::parse(HEIGHT_DOC).unwrap(); - let loader = move |_: &str, _: Option<&str>| { - Some(TextureRef(std::sync::Arc::new(Ramp { slope_u, slope_v }))) - }; - let mut c = Compiler::new(&doc, &Host::new(&loader)); - let slot = c.compile_named("", node, None); - let mut slots = Vec::new(); - let ctx = ShadeCtx { - uv: (0.3, 0.6), - normal: Vec3A::Z, - tangent: Vec3A::X, - view: -Vec3A::Z, - position: Vec3A::ZERO, - uv_width, - }; - c.program.eval(&ctx, &mut slots); - slots[slot as usize].rgb() - } - - #[test] - fn heighttonormal_is_the_osl_reference_over_one_footprint() { - // A height rising 10 per UV unit, over a 0.01-wide footprint, changes - // by 0.1 across it: the reference's `Dx(in)`. - let (du, dv) = (0.1f32, 0.0f32); - let dz = (1.0 - du * du - dv * dv).sqrt(); - let want = Vec3A::new(-du, -dv, dz).normalize() * 0.5 + Vec3A::splat(0.5); - close(height_to_normal_at(10.0, 0.0, 0.01, "n"), want); - let want = Vec3A::new(0.0, -0.1, dz).normalize() * 0.5 + Vec3A::splat(0.5); - close(height_to_normal_at(0.0, 10.0, 0.01, "n"), want); - } - - #[test] - fn heighttonormal_tilts_away_from_rising_height() { - // Through `normalmap` in a frame with the tangent along +u: a height - // rising along +u (+v) is a slope facing −u (−v), which is where the - // normal must lean. - let n = height_to_normal_at(10.0, 0.0, 0.01, "world"); - assert!(n.x < -0.05 && n.y.abs() < 1e-6 && n.z > 0.9, "{n}"); - let n = height_to_normal_at(0.0, 10.0, 0.01, "world"); - assert!(n.y < -0.05 && n.x.abs() < 1e-6 && n.z > 0.9, "{n}"); - } - - #[test] - fn heighttonormal_without_a_footprint_is_flat() { - // No ray cone, no derivative — the nodedef's own default output. - assert_eq!( - height_to_normal_at(10.0, 5.0, 0.0, "n"), - Vec3A::new(0.5, 0.5, 1.0) - ); - } - - /// `node` of `doc`, every image in it served by `tex`, at uv (0.3, 0.6) - /// with a footprint `uv_width` wide. Also returns the compiled program. - fn eval_with( - doc: &str, - node: &str, - tex: impl crate::Texture + Clone + 'static, - uv_width: f32, - ) -> (Val, Program) { - let doc = Doc::parse(doc).unwrap(); - let loader = - move |_: &str, _: Option<&str>| Some(TextureRef(std::sync::Arc::new(tex.clone()))); - let mut c = Compiler::new(&doc, &Host::new(&loader)); - let slot = c.compile_named("", node, None); - let mut slots = Vec::new(); - let ctx = ShadeCtx { - uv: (0.3, 0.6), - normal: Vec3A::Z, - tangent: Vec3A::X, - view: -Vec3A::Z, - position: Vec3A::ZERO, - uv_width, - }; - c.program.eval(&ctx, &mut slots); - (slots[slot as usize], c.program) - } - - #[derive(Clone)] - struct RampTex(f32); - impl crate::Texture for RampTex { - fn eval(&self, u: f32, _: f32, _: f32) -> [f32; 4] { - [self.0 * u; 4] - } - } - - /// `heighttonormal` over an image whose `texcoord` is `coord` — a - /// fragment of nodes naming the connected one `c`, or empty for none. - fn height_doc(coord: &str) -> String { - let conn = if coord.is_empty() { - String::new() - } else { - r#""#.into() - }; - format!( - r#" - {coord} - - - {conn} - - - - - "# - ) - } - - fn texture_coords(p: &Program) -> Vec> { - p.ops - .iter() - .filter_map(|op| match op { - Op::Texture { coord, .. } => Some(*coord), - _ => None, - }) - .collect() - } - - #[test] - fn heighttonormal_of_a_constant_coordinate_is_flat() { - // Every tap reads the same texel, whatever the footprint. - let doc = height_doc( - r#""#, - ); - let (n, _) = eval_with(&doc, "n", RampTex(10.0), 0.01); - assert_eq!(n.rgb(), Vec3A::new(0.5, 0.5, 1.0)); - } - - #[test] - fn the_default_chart_spelled_out_is_the_implicit_one() { - // `geompropvalue st` and `texcoord` are what `ctx.uv` already holds: - // same answer as no connection, and still the JIT's inline lookup. - let (want, p) = eval_with(&height_doc(""), "n", RampTex(10.0), 0.01); - assert!(texture_coords(&p).iter().all(Option::is_none)); - for c in [ - r#""#, - r#""#, - ] { - let (got, p) = eval_with(&height_doc(c), "n", RampTex(10.0), 0.01); - assert_eq!(bits_of(got), bits_of(want), "{c}"); - assert!(texture_coords(&p).iter().all(Option::is_none), "{c}"); - } - } - - fn bits_of(v: Val) -> ([u32; 4], u8) { - (v.v.map(f32::to_bits), v.arity) - } - - #[test] - fn an_authored_coordinate_is_sampled_and_differentiated_through() { - // `texcoord · 2`: the image reads at twice the chart, and the height - // changes twice as fast across the same footprint. - let doc = height_doc( - r#" - - - - "#, - ); - let (h, _) = eval_with(&doc, "h", RampTex(10.0), 0.01); - assert!((h.x() - 10.0 * 0.6).abs() < 1e-5, "height {}", h.x()); - let (n, _) = eval_with(&doc, "n", RampTex(10.0), 0.01); - let du = 0.2f32; // 10 per unit, 2x the chart, 0.01 across - let want = - Vec3A::new(-du, 0.0, (1.0 - du * du).sqrt()).normalize() * 0.5 + Vec3A::splat(0.5); - close(n.rgb(), want); - } - - #[test] - fn an_uncompilable_coordinate_falls_back_to_the_chart() { - // `place2d` has no operator here; its constant stand-in would pin the - // lookup to one texel, so the chart is used instead (and reported). - let doc = height_doc(r#""#); - let (h, p) = eval_with(&doc, "h", RampTex(10.0), 0.01); - assert!((h.x() - 3.0).abs() < 1e-5, "height {}", h.x()); - assert!(texture_coords(&p).iter().all(Option::is_none)); - } - - #[derive(Clone)] - struct Channels; - impl crate::Texture for Channels { - fn eval(&self, _: f32, _: f32, _: f32) -> [f32; 4] { - [0.25, 0.5, 0.75, 1.0] - } - } - - #[test] - fn colorcorrect_of_a_float_image_is_grey() { - // A one-lane lookup keeps its file's other channels in lanes 1..3; the - // correction must promote lane 0, not correct and expose the rest. - let doc = r#" - - - - - - - - - - - "#; - for node in ["cc", "out"] { - let (v, _) = eval_with(doc, node, Channels, 0.0); - assert_eq!(v.arity, 3, "{node}"); - assert_eq!(v.rgb(), Vec3A::splat(0.5), "{node}"); - } - } - - #[test] - fn heighttonormal_asks_the_host_for_its_image_once() { - // Four shifted copies of the lookup, one sampler: the host's decode - // and residency are per file, not per tap. - let doc = Doc::parse(HEIGHT_DOC).unwrap(); - let calls = std::cell::Cell::new(0); - let loader = |_: &str, _: Option<&str>| { - calls.set(calls.get() + 1); - Some(TextureRef(std::sync::Arc::new(Ramp { - slope_u: 1.0, - slope_v: 0.0, - }))) - }; - let mut c = Compiler::new(&doc, &Host::new(&loader)); - c.compile_named("", "n", None); - c.compile_named("", "h", None); - assert_eq!(calls.get(), 1); - } - - #[test] - fn division_by_zero_stays_finite() { - // `1 / transmittance` with a black channel is authored in the teapot's - // own ceramic graph; an infinity there survives every later multiply - // and reaches the framebuffer as a NaN pixel. - let v = run( - r#" - - - - - "#, - "d", - ); - assert!(v.x().is_finite()); - } -} diff --git a/crates/crust-mtlx/src/eval/apply.rs b/crates/crust-mtlx/src/eval/apply.rs new file mode 100644 index 00000000..928b0543 --- /dev/null +++ b/crates/crust-mtlx/src/eval/apply.rs @@ -0,0 +1,517 @@ +//! The interpreter: one [`Op`] applied to the slots computed so far, and the +//! math its operators share. + +use glam::Vec3A; + +use super::{BinOp, Op, ShadeCtx, UnOp}; +use crate::value::Val; + +/// One instruction's value, given the slots computed before it. +/// +/// Shared by [`Program::eval`] and the constant folder in +/// [`Program::optimize`], which is what makes a folded constant exactly the +/// value the interpreter would have produced. +#[inline(always)] +pub(super) fn apply(op: &Op, slots: &[Val], ctx: &ShadeCtx) -> Val { + // Every operand index was emitted before this instruction, so the slot + // exists. `get` rather than indexing keeps a malformed program from + // panicking inside the integrator. + let g = |i: u32| -> Val { slots.get(i as usize).copied().unwrap_or(Val::ZERO) }; + match op { + Op::Const(v) => *v, + Op::Texture { + tex, + fallback, + scale, + offset, + arity, + shift, + coord, + } => match tex { + Some(t) => { + let (u, v) = match coord { + Some(c) => { + let c = g(*c); + // A `float` coordinate is both axes, as MaterialX's + // implicit promotion to `vector2` makes it. + if c.arity == 1 { + (c.x(), c.x()) + } else { + (c.v[0], c.v[1]) + } + } + None => shifted_uv(ctx, *shift), + }; + let u = u * scale[0] + offset[0]; + let v = v * scale[1] + offset[1]; + // `uvtiling` scales the coordinates, so it scales the + // footprint with them: a texture tiled 10× is being + // minified 10× and must read a coarser level to match. + // The two axes are averaged because the width is one + // isotropic number. + let w = ctx.uv_width * 0.5 * (scale[0].abs() + scale[1].abs()); + let rgba = t.eval(u, v, w); + Val { + v: rgba, + arity: *arity, + } + } + None => *fallback, + }, + Op::TexCoord { shift } => { + let (u, v) = shifted_uv(ctx, *shift); + Val::vec2(u, v) + } + Op::Normal => ctx.normal.into(), + Op::ViewDirection => ctx.view.into(), + Op::Position => ctx.position.into(), + Op::Unary { op, a } => { + let a = g(*a); + match op { + UnOp::Abs => a.map(f32::abs), + // Guarded so a zero or negative operand — which the + // teapot's Beer-Lambert chain can produce from a + // black texel — yields a finite value instead of an + // infinity that then poisons every downstream lane. + // The guard is OSL's `safe_log`, which the MaterialX + // reference runs: the operand is raised to the smallest + // normal float, so `ln(0)` is `ln(f32::MIN_POSITIVE)`. + UnOp::Ln => a.map(|x| x.max(f32::MIN_POSITIVE).ln()), + UnOp::Exp => a.map(|x| x.clamp(-88.0, 88.0).exp()), + UnOp::Sin => a.map(f32::sin), + UnOp::Cos => a.map(f32::cos), + UnOp::Asin => a.map(|x| x.clamp(-1.0, 1.0).asin()), + UnOp::Acos => a.map(|x| x.clamp(-1.0, 1.0).acos()), + UnOp::Sqrt => a.map(|x| x.max(0.0).sqrt()), + // Not `f32::signum`, which answers ±1 for ±0: MaterialX's + // `sign` of zero is zero. + UnOp::Sign => a.map(|x| { + if x > 0.0 { + 1.0 + } else if x < 0.0 { + -1.0 + } else { + x + } + }), + UnOp::Floor => a.map(f32::floor), + UnOp::Ceil => a.map(f32::ceil), + UnOp::Normalize => normalize(a), + } + } + Op::Binary { op, a, b } => { + let (a, b) = (g(*a), g(*b)); + match op { + BinOp::Add => a + b, + BinOp::Sub => a - b, + BinOp::Mul => a * b, + // A zero divisor is a real possibility in these + // graphs (`1 / transmittance` with a black channel), + // and an infinity survives every later multiply. + BinOp::Div => a.zip(b, |x, y| if y.abs() > 1e-20 { x / y } else { 0.0 }), + BinOp::Pow => a.zip(b, safe_pow), + BinOp::Min => a.zip(b, f32::min), + BinOp::Max => a.zip(b, f32::max), + BinOp::Modulo => a.zip(b, floored_mod), + } + } + // `bg·(1−m) + fg·m`. Written as two weighted terms rather + // than `bg + (fg−bg)·m` so that a `float` mix against wider + // operands broadcasts through `zip`'s promotion in both + // terms alike. + Op::Mix { fg, bg, m } => { + let (fg, bg, m) = (g(*fg), g(*bg), g(*m)); + let inv = m.map(|x| 1.0 - x); + bg * inv + fg * m + } + // `max(min(in, high), low)`, OSL's order: it only shows when + // `low > high`, where `low` wins. + Op::Clamp { a, low, high } => { + let (a, lo, hi) = (g(*a), g(*low), g(*high)); + a.zip(hi, f32::min).zip(lo, f32::max) + } + Op::Contrast { a, amount, pivot } => { + let (a, amt, piv) = (g(*a), g(*amount), g(*pivot)); + (a - piv) * amt + piv + } + Op::Remap { + a, + in_low, + in_high, + out_low, + out_high, + } => { + let (a, il, ih, ol, oh) = (g(*a), g(*in_low), g(*in_high), g(*out_low), g(*out_high)); + let t = (a - il).zip(ih - il, |x, d| if d.abs() > 1e-20 { x / d } else { 0.0 }); + ol + (oh - ol) * t + } + Op::Invert { a, amount } => g(*amount) - g(*a), + Op::Convert { a, arity } => convert(g(*a), *arity), + Op::Extract { a, index } => Val::float(g(*a).v[(*index).min(3)]), + Op::Combine3 { a, b, c } => Val::vec3(g(*a).x(), g(*b).x(), g(*c).x()), + Op::Combine2 { a, b } => combine2(g(*a), g(*b)), + Op::DotProduct { a, b } => dot(g(*a), g(*b)), + Op::Luminance { a, coeffs } => { + let a = g(*a); + let l = a.rgb().dot(g(*coeffs).rgb()); + // The input's width: `color3` is the grey `(l, l, l)`, and + // `color4` keeps its alpha. + if a.arity == 4 { + Val::vec4(l, l, l, a.v[3]) + } else { + Val::float(l).broadcast_to(a.arity) + } + } + Op::NormalMap { a, scale } => normal_map(g(*a), g(*scale), ctx).into(), + Op::ArtisticIor { + reflectivity, + edge, + extinction, + } => { + let (n, k) = artistic_ior(g(*reflectivity).rgb(), g(*edge).rgb()); + if *extinction { k.into() } else { n.into() } + } + Op::Smoothstep { a, low, high } => { + let (a, lo, hi) = (g(*a), g(*low), g(*high)); + let n = a.arity.max(lo.arity).max(hi.arity); + let (a, lo, hi) = (a.broadcast_to(n), lo.broadcast_to(n), hi.broadcast_to(n)); + Val { + v: [0, 1, 2, 3].map(|i| smoothstep(a.v[i], lo.v[i], hi.v[i])), + arity: n, + } + } + Op::HsvAdjust { a, amount } => { + let hsv = rgb_to_hsv(g(*a).rgb()); + let m = g(*amount).rgb(); + hsv_to_rgb(Vec3A::new(hsv.x + m.x, hsv.y * m.y, hsv.z * m.z)).into() + } + Op::HeightToNormal { + xp, + xm, + yp, + ym, + scale, + } => height_to_normal( + g(*xp).x() - g(*xm).x(), + g(*yp).x() - g(*ym).x(), + g(*scale).x(), + ) + .into(), + } +} + +/// The chart coordinates `shift` footprint widths away from the shading +/// point. +/// +/// A zero shift returns `ctx.uv` untouched rather than adding `0 · width`, +/// which keeps every ordinary lookup bit-identical to what it was before +/// shifts existed (and to the JIT's inline texture path, which never sees a +/// shifted op). A shift over a zero footprint — no ray cone — lands back on +/// the shading point, so a derivative taken from shifted copies reads zero. +#[inline] +pub(super) fn shifted_uv(ctx: &ShadeCtx, shift: [f32; 2]) -> (f32, f32) { + if shift == [0.0, 0.0] { + ctx.uv + } else { + ( + ctx.uv.0 + shift[0] * ctx.uv_width, + ctx.uv.1 + shift[1] * ctx.uv_width, + ) + } +} + +/// MaterialX's OSL `mx_heighttonormal_vector3`, given the height's change +/// across one footprint along `u` (`du`) and `v` (`dv`). +/// +/// The reference reads `dx = -Dx(in)`, `dy = Dy(in)`: screen-space +/// derivatives, i.e. the height's change across one pixel. The footprint is +/// this renderer's pixel (a ray cone's width in chart units), so the change +/// across it is the same quantity, with `Dx` running along `u`. Raster `y` +/// runs *down* while `v` runs up, so `Dy(in) = -dv` and both lateral +/// components come out as `-dh`: the normal of a surface raised by `in`. +/// That also makes the result resolution-dependent exactly as the +/// reference's is — a bump reads steeper the coarser the footprint. +pub(super) fn height_to_normal(du: f32, dv: f32, scale: f32) -> Vec3A { + let (dx, dy) = (-du, -dv); + let dz = scale.max(1.0e-5) * (1.0 - dx * dx - dy * dy).max(1.0e-5).sqrt(); + Vec3A::new(dx, dy, dz).normalize_or(Vec3A::Z) * 0.5 + Vec3A::splat(0.5) +} + +/// MaterialX's `mx_rgbtohsv` (Foley & van Dam, via OSL), transcribed. +pub(super) fn rgb_to_hsv(c: Vec3A) -> Vec3A { + let (r, g, b) = (c.x, c.y, c.z); + let min = r.min(g.min(b)); + let max = r.max(g.max(b)); + let delta = max - min; + let s = if max > 0.0 { delta / max } else { 0.0 }; + let h = if s <= 0.0 { + 0.0 + } else { + let h = if r >= max { + (g - b) / delta + } else if g >= max { + 2.0 + (b - r) / delta + } else { + 4.0 + (r - g) / delta + } * (1.0 / 6.0); + if h < 0.0 { h + 1.0 } else { h } + }; + Vec3A::new(h, s, max) +} + +/// MaterialX's `mx_hsvtorgb`, transcribed. The hue wraps, so a hue shift +/// past 1 comes round again. +pub(super) fn hsv_to_rgb(hsv: Vec3A) -> Vec3A { + let (h, s, v) = (hsv.x, hsv.y, hsv.z); + if s < 0.0001 { + return Vec3A::splat(v); + } + let h = 6.0 * (h - h.floor()); + // `h` is in [0, 6) up to rounding; a non-finite hue lands in the last + // sextant rather than anywhere undefined. + let hi = h.trunc(); + let f = h - hi; + let p = v * (1.0 - s); + let q = v * (1.0 - s * f); + let t = v * (1.0 - s * (1.0 - f)); + match hi as i32 { + 0 => Vec3A::new(v, t, p), + 1 => Vec3A::new(q, v, p), + 2 => Vec3A::new(p, v, t), + 3 => Vec3A::new(p, q, v), + 4 => Vec3A::new(t, p, v), + _ => Vec3A::new(v, p, q), + } +} + +/// MaterialX `normalize`, over the value's own lanes. A zero-length input is +/// returned unchanged rather than divided into NaNs. A `float` broadcasts to a +/// `vector3`, as it always did here (MaterialX has no `float` variant). +pub(super) fn normalize(a: Val) -> Val { + match a.arity { + 4 => { + let v = glam::Vec4::from_array(a.v); + let n = v.length(); + if n > 1e-20 { + let [x, y, z, w] = (v / n).to_array(); + Val::vec4(x, y, z, w) + } else { + a + } + } + arity => { + // Lanes past a `vector2`'s are whatever the op that made it left + // there, so they are zeroed before they can enter the length. + let v = if arity == 2 { + Vec3A::new(a.v[0], a.v[1], 0.0) + } else { + a.rgb() + }; + let n = v.length(); + if n > 1e-20 { + let r = v / n; + if arity == 2 { + Val::vec2(r.x, r.y) + } else { + r.into() + } + } else { + a + } + } + } +} + +/// MaterialX `dotproduct` over the operands' lanes. The `vector3` sum is +/// glam's, as it always was; the others extend it. (A `float` operand's +/// lanes all hold its value, so reading them broadcasts it.) +pub(super) fn dot(a: Val, b: Val) -> Val { + let n = a.arity.max(b.arity); + let lanes = |v: Val| match n { + // A `vector2`'s third lane is not its own; see `normalize`. + 2 => Vec3A::new(v.v[0], v.v[1], 0.0), + _ => v.rgb(), + }; + let d = lanes(a).dot(lanes(b)); + Val::float(if n == 4 { d + a.v[3] * b.v[3] } else { d }) +} + +/// MaterialX `convert`. A `float` broadcasts. Widening a wider value fills +/// the new lanes the way the nodedefs do — zero, except that a `color4` / +/// `vector4` made from fewer lanes gets `1` in its last (an opaque alpha). +/// Narrowing keeps the leading lanes; to a `float`, the first. +pub(super) fn convert(a: Val, arity: u8) -> Val { + let arity = arity.clamp(1, 4); + if a.arity == 1 { + return a.with_arity(arity); + } + if arity == 1 { + // Every lane of a `float` holds its value; see `Val::float`. + return Val::float(a.v[0]); + } + let mut v = a.v; + for (i, lane) in v.iter_mut().enumerate().skip(a.arity as usize) { + *lane = if i == 3 { 1.0 } else { 0.0 }; + } + Val { v, arity } +} + +/// MaterialX `combine2`: the lanes of `a`, then of `b`. Covers every +/// signature — `(float, float)` → `vector2`, `(color3, float)` → `color4`, +/// `(vector3, float)` and `(vector2, vector2)` → `vector4`. +pub(super) fn combine2(a: Val, b: Val) -> Val { + let mut v = [0.0; 4]; + let (na, nb) = (a.arity as usize, b.arity as usize); + v[..na].copy_from_slice(&a.v[..na]); + let nb = nb.min(4 - na.min(4)); + v[na..na + nb].copy_from_slice(&b.v[..nb]); + Val { + v, + arity: (na + nb) as u8, + } +} + +/// MaterialX's `modulo`, OSL's `mod`: floored, so the result takes the +/// divisor's sign (`-0.2 mod 1` is `0.8`) where Rust's `%` truncates, and a +/// zero divisor returns the dividend, as OSL's does. +/// +/// OSL's `x − y·floor(x / y)` is kept wherever its quotient is finite, so +/// the rounding matches the reference (at `-1 mod -0.2` the quotient rounds +/// to 5 and the result to 0, a period away from the exact −0.19999999). Its +/// quotient overflows for some finite operands, though (`1 mod 1e-40` would +/// be `−inf`), and there the exact remainder `%` takes over, moved by one `y` +/// when its sign is the dividend's rather than the divisor's. +pub(super) fn floored_mod(x: f32, y: f32) -> f32 { + if y == 0.0 { + return x; + } + let q = (x / y).floor(); + if q.is_finite() { + return x - y * q; + } + let r = x % y; + if r != 0.0 && (r < 0.0) != (y < 0.0) { + r + y + } else { + r + } +} + +/// OSL's `pow` (OIIO `safe_pow`), which MaterialX's `power` is: `x^0` is one, +/// `0^y` zero, a negative base takes only integer exponents (zero otherwise), +/// and the result is clamped finite. +pub(super) fn safe_pow(x: f32, y: f32) -> f32 { + if y == 0.0 { + return 1.0; + } + if x == 0.0 { + return 0.0; + } + if x < 0.0 && y != y.floor() { + return 0.0; + } + x.powf(y).clamp(-f32::MAX, f32::MAX) +} + +/// OSL's `smoothstep(low, high, x)`, which MaterialX's is: zero below `low`, +/// one from `high` up, the Hermite ramp between. With `low >= high` the first +/// two tests decide every `x`, so no division by the empty interval happens. +pub(super) fn smoothstep(x: f32, low: f32, high: f32) -> f32 { + if x < low { + 0.0 + } else if x >= high { + 1.0 + } else { + let t = (x - low) / (high - low); + t * t * (3.0 - 2.0 * t) + } +} + +/// MaterialX `normalmap`: decode `[0,1]`-encoded tangent-space vector, scale +/// its lateral components, and rotate it into world space. +/// +/// Falls back to the geometric normal when there is no tangent — the host +/// says when that happens (in crust, only baked single-placement geometry +/// carries one; see crust-core's `UvMap::tangents`). Returning the +/// geometric normal is the right degradation: a normal map's *mean* is the +/// surface normal, so the flat surface is the map's own zero. +pub(super) fn normal_map(encoded: Val, scale: Val, ctx: &ShadeCtx) -> Vec3A { + let v = encoded.rgb() * 2.0 - Vec3A::ONE; + // `scale` is a `float` or, per axis, a `vector2`. + let (sx, sy) = if scale.arity >= 2 { + (scale.v[0], scale.v[1]) + } else { + (scale.x(), scale.x()) + }; + let finite = |s: f32| if s.is_finite() { s } else { 1.0 }; + let local = Vec3A::new(v.x * finite(sx), v.y * finite(sy), v.z.max(1e-4)); + perturb_normal(local, ctx.normal, ctx.tangent) +} + +/// Rotates a **decoded** tangent-space normal (`z` along `normal`) into world +/// space, against `tangent` re-orthogonalised to `normal`. +/// +/// The half of `normal_map` that knows nothing about MaterialX's `[0,1]` +/// encoding, public so a host with its own decode — UsdPreviewSurface's +/// `normal` input arrives already in `[-1,1]`, its UsdUVTexture's +/// `scale`/`bias` having done the decode — rotates it identically. Returns +/// `normal` unchanged when `tangent` is zero (no chart frame) or parallel to +/// it, for the reason `normal_map` gives. +pub fn perturb_normal(local: Vec3A, normal: Vec3A, tangent: Vec3A) -> Vec3A { + if tangent.length_squared() < 1e-20 { + return normal; + } + let n = normal; + // Re-orthogonalise: the stored tangent is the triangle's, while `n` may + // already carry interpolated shading curvature, so the two need not be + // perpendicular. + let t = (tangent - n * n.dot(tangent)).normalize_or_zero(); + if t.length_squared() < 1e-20 { + return n; + } + let b = n.cross(t); + let world = t * local.x + b * local.y + n * local.z; + if world.length_squared() > 1e-20 { + world.normalize() + } else { + n + } +} + +/// Gulbrandsen's "Artist Friendly Metallic Fresnel": normal-incidence +/// reflectivity and grazing edge tint → complex IOR. +/// +/// Implemented rather than short-circuited (the conductor lobe wants a +/// reflectivity back, which is what went in) because a graph may author `ior` +/// and `extinction` directly, and the round trip through +/// [`reflectivity_from_ior`] then handles both authorings with one path. +pub(super) fn artistic_ior(reflectivity: Vec3A, edge: Vec3A) -> (Vec3A, Vec3A) { + let r = reflectivity.clamp(Vec3A::ZERO, Vec3A::splat(0.99)); + let rs = Vec3A::new(r.x.sqrt(), r.y.sqrt(), r.z.sqrt()); + let n_min = (Vec3A::ONE - r) / (Vec3A::ONE + r); + let n_max = (Vec3A::ONE + rs) / (Vec3A::ONE - rs).max(Vec3A::splat(1e-6)); + // OSL's `mix(n_max, n_min, edge)`, as `x·(1 − t) + y·t`: an edge colour + // of white — the default — selects `n_min` exactly, where `x + (y − x)·t` + // would leave `n_max`'s rounding in it (n_max is ~70 for a bright metal). + // Clamped, unlike the reference: an edge tint outside [0, 1] extrapolates + // the IOR to nonsense, down to negative values. + let e = edge.clamp(Vec3A::ZERO, Vec3A::ONE); + let n = n_max * (Vec3A::ONE - e) + n_min * e; + let np1 = n + Vec3A::ONE; + let nm1 = n - Vec3A::ONE; + let k2 = + ((np1 * np1 * r - nm1 * nm1) / (Vec3A::ONE - r).max(Vec3A::splat(1e-6))).max(Vec3A::ZERO); + (n, Vec3A::new(k2.x.sqrt(), k2.y.sqrt(), k2.z.sqrt())) +} + +/// Normal-incidence reflectivity of a conductor with complex IOR `n + ik`. +/// +/// The exact inverse of `artistic_ior` (the node's implementation above), which +/// is what lets a conductor lobe +/// be reduced to the one colour OpenPBR's metal lobe takes, whichever way the +/// graph authored it. +pub fn reflectivity_from_ior(n: Vec3A, k: Vec3A) -> Vec3A { + let num = (n - Vec3A::ONE) * (n - Vec3A::ONE) + k * k; + let den = ((n + Vec3A::ONE) * (n + Vec3A::ONE) + k * k).max(Vec3A::splat(1e-6)); + (num / den).clamp(Vec3A::ZERO, Vec3A::ONE) +} diff --git a/crates/crust-mtlx/src/eval/compile.rs b/crates/crust-mtlx/src/eval/compile.rs new file mode 100644 index 00000000..056812eb --- /dev/null +++ b/crates/crust-mtlx/src/eval/compile.rs @@ -0,0 +1,680 @@ +//! The compiler: named `.mtlx` nodes turned into a topologically ordered +//! [`Program`]. + +use glam::Vec3A; + +use super::*; +use crate::parse::{Doc, Input, Node, Source}; +use crate::texture::TextureRef; +use crate::value::{Val, arity_of, convert_color, is_color_type}; + +/// `luminance`'s default `lumacoeffs`: ACEScg's (AP1) weights. +const AP1_LUMA_COEFFS: Val = Val::vec3(0.2722287, 0.6740818, 0.0536895); + +/// Turns named `.mtlx` nodes into a topologically ordered [`Program`]. +pub struct Compiler<'a> { + pub doc: &'a Doc, + pub program: Program, + /// Slot already emitted for a `(graph, node, output, shift)` key, so a + /// node feeding five others is evaluated once. The output name matters: + /// `artistic_ior` emits a different slot for `ior` than for `extinction`. + /// So does the shift: a node under `heighttonormal` is compiled once per + /// offset it is differentiated at. + memo: std::collections::HashMap<(String, String, String, [u32; 2]), u32>, + /// The offset, in footprint widths, every texture lookup and `texcoord` + /// compiled now is displaced by — zero except while `heighttonormal` + /// compiles the shifted copies of its input. + uv_shift: [f32; 2], + /// What the loader answered per `(file, colorspace)` — the space as + /// handed to it — so the shifted copies of an `image`, and two images of + /// one file in one effective space, share the one sampler rather than + /// asking the host again. + images: std::collections::HashMap<(String, Option), Option>, + /// Nodes currently being compiled, so a cyclic document — which a + /// hand-edited `.mtlx` can be — terminates as a constant rather than + /// recursing until the stack runs out. + active: Vec, + /// Resolves an `image` node's `file` input to a sampler, and converts + /// authored colours into the working space. + host: crate::Host<'a>, + /// Node categories met that this compiler has no operator for, for one + /// summary warning instead of one per occurrence. + pub unsupported: std::collections::BTreeSet, +} + +impl<'a> Compiler<'a> { + pub fn new(doc: &'a Doc, host: &crate::Host<'a>) -> Compiler<'a> { + Compiler { + doc, + program: Program::default(), + memo: std::collections::HashMap::new(), + uv_shift: [0.0, 0.0], + images: std::collections::HashMap::new(), + active: Vec::new(), + host: *host, + unsupported: Default::default(), + } + } + + pub fn emit(&mut self, op: Op) -> u32 { + self.program.ops.push(op); + (self.program.ops.len() - 1) as u32 + } + + pub fn constant(&mut self, v: Val) -> u32 { + self.emit(Op::Const(v)) + } + + /// `luminance` at the nodedef's default `lumacoeffs`, ACEScg's (AP1) — + /// what the stdlib nodegraphs' unauthored `luminance` / `saturate` read. + pub fn luminance(&mut self, a: u32) -> u32 { + let coeffs = self.constant(AP1_LUMA_COEFFS); + self.emit(Op::Luminance { a, coeffs }) + } + + /// The value of `slot` when it is a compile-time constant — a literal, or + /// a pure operator over constants — and `None` when it depends on the + /// shading point or a texture. + /// + /// What the closure builders prune on: a branch whose weight folds to a + /// literal zero, or a mix whose factor folds to exactly 0 or 1, is left + /// out of the tree altogether rather than evaluated at every vertex to be + /// told it contributes nothing. + pub fn fold(&self, slot: u32) -> Option { + let ops = &self.program.ops; + let op = ops.get(slot as usize)?; + if let Op::Const(v) = op { + return Some(*v); + } + if !op.is_pure() { + return None; + } + let mut operands = Vec::new(); + op.clone().for_each_operand(|o| operands.push(*o)); + let mut slots = vec![Val::ZERO; slot as usize]; + for o in operands { + slots[o as usize] = self.fold(o)?; + } + let ctx = ShadeCtx { + uv: (0.0, 0.0), + normal: Vec3A::Z, + tangent: Vec3A::X, + view: -Vec3A::Z, + position: Vec3A::ZERO, + uv_width: 0.0, + }; + Some(apply(op, &slots, &ctx)) + } + + /// Compiles the value feeding `input` of `node`, or `default` when the + /// input is unauthored. + pub fn input_or(&mut self, node: &Node, name: &str, default: Val) -> u32 { + match node.input(name) { + Some(i) => self.compile_input(node, i), + // The nodedef's default is typed: an unauthored `in1` of an + // `add_color3` is a colour3 zero, not a `float` one, and an op over + // nothing but defaults must come out at the node's width. A + // `float` default's lanes already hold its value, so widening it + // changes no lane — only what a width-reading consumer + // (`luminance`, `normalize`, `dotproduct`, `combine2`) sees. Not + // `convert`, whose input width is what decides the alpha. + None if default.arity == 1 && node.category != "convert" => { + self.constant(default.broadcast_to(arity_of(&node.type_name))) + } + None => self.constant(default), + } + } + + /// Like [`Compiler::input_or`] but reports whether the input was authored, + /// which the BSDF reducer needs to tell "no normal map" from "a normal map + /// that happens to be flat". + pub fn optional_input(&mut self, node: &Node, name: &str) -> Option { + let i = node.input(name)?; + Some(self.compile_input(node, i)) + } + + /// The width an authored input carries: its producer's declared type + /// when it is connected, since the parser reads an input with no `type` + /// attribute as a `float` whatever feeds it, and the input's own type for + /// a literal. A multioutput producer's outputs are not typed in the + /// document, so there the input's declaration is all there is. + fn input_arity(&self, node: &Node, input: &Input) -> u8 { + let scope = node.graph.clone().unwrap_or_default(); + let producer = match &input.source { + Source::Node { name, .. } => self.doc.find(&scope, name), + Source::Graph { graph, output } => self.doc.graph_output(graph, output).map(|c| c.node), + Source::Value(_) => None, + }; + match producer { + Some(p) if p.type_name != "multioutput" => arity_of(&p.type_name), + _ => arity_of(&input.type_name), + } + } + + fn compile_input(&mut self, node: &Node, input: &Input) -> u32 { + let scope = node.graph.clone().unwrap_or_default(); + match &input.source { + Source::Value(v) => { + // A literal "0.5" on a color3 input broadcasts, which the + // parser already arranged; a colour is then taken into the + // working space. + let v = self.managed(node, input, *v); + self.constant(v) + } + Source::Node { name, output } => self.compile_named(&scope, name, output.as_deref()), + Source::Graph { graph, output } => match self.doc.graph_output(graph, output) { + Some(conn) => { + // The graph's `` may itself select one output of a + // multioutput node; carry it through, or `extinction` + // silently compiles as `ior`. + let (g, nm) = ( + conn.node.graph.clone().unwrap_or_default(), + conn.node.name.clone(), + ); + let sel = conn.output.map(str::to_string); + self.compile_named(&g, &nm, sel.as_deref()) + } + None => self.constant(Val::ZERO), + }, + } + } + + /// Compiles the node called `name`, returning its slot. + pub fn compile_named(&mut self, scope: &str, name: &str, output: Option<&str>) -> u32 { + let key = ( + scope.to_string(), + name.to_string(), + output.unwrap_or("").to_string(), + self.uv_shift.map(f32::to_bits), + ); + if let Some(&slot) = self.memo.get(&key) { + return slot; + } + if self.active.iter().any(|a| a == name) { + // A cycle. Break it with a constant — the alternative is a stack + // overflow at import on a malformed document. + return self.constant(Val::ZERO); + } + let Some(node) = self.doc.find(scope, name) else { + return self.constant(Val::ZERO); + }; + let node = node.clone(); + self.active.push(name.to_string()); + let slot = self.compile_node(&node, output); + self.active.pop(); + self.memo.insert(key, slot); + slot + } + + fn compile_node(&mut self, node: &Node, output: Option<&str>) -> u32 { + let arity = arity_of(&node.type_name); + let bin = |c: &mut Self, op: BinOp, d1: Val, d2: Val| { + let a = c.input_or(node, "in1", d1); + let b = c.input_or(node, "in2", d2); + c.emit(Op::Binary { op, a, b }) + }; + let un = |c: &mut Self, op: UnOp| { + let a = c.input_or(node, "in", Val::ZERO); + c.emit(Op::Unary { op, a }) + }; + match node.category.as_str() { + "constant" => self.input_or(node, "value", Val::ZERO), + "image" | "tiledimage" => self.compile_image(node, arity), + "texcoord" => self.emit(Op::TexCoord { + shift: self.uv_shift, + }), + "normal" => self.emit(Op::Normal), + "viewdirection" => self.emit(Op::ViewDirection), + "position" => self.emit(Op::Position), + "add" => bin(self, BinOp::Add, Val::ZERO, Val::ZERO), + "subtract" => bin(self, BinOp::Sub, Val::ZERO, Val::ZERO), + // The unauthored defaults are the nodedefs': `in1` is zero and + // `in2` one for every operator whose identity is one. + "multiply" => bin(self, BinOp::Mul, Val::ZERO, Val::ONE), + "divide" => bin(self, BinOp::Div, Val::ZERO, Val::ONE), + "power" => bin(self, BinOp::Pow, Val::ZERO, Val::ONE), + "min" => bin(self, BinOp::Min, Val::ZERO, Val::ZERO), + "max" => bin(self, BinOp::Max, Val::ZERO, Val::ZERO), + "modulo" => bin(self, BinOp::Modulo, Val::ZERO, Val::ONE), + "absval" => un(self, UnOp::Abs), + "ln" => { + let a = self.input_or(node, "in", Val::ONE); + self.emit(Op::Unary { op: UnOp::Ln, a }) + } + "exp" => un(self, UnOp::Exp), + "sin" => un(self, UnOp::Sin), + "cos" => un(self, UnOp::Cos), + "asin" => un(self, UnOp::Asin), + "acos" => un(self, UnOp::Acos), + "sqrt" => un(self, UnOp::Sqrt), + "sign" => un(self, UnOp::Sign), + "floor" => un(self, UnOp::Floor), + "ceil" => un(self, UnOp::Ceil), + "normalize" => un(self, UnOp::Normalize), + "luminance" => { + let a = self.input_or(node, "in", Val::ZERO); + // The nodedef's default: ACEScg's (AP1) coefficients. + let coeffs = self.input_or(node, "lumacoeffs", AP1_LUMA_COEFFS); + self.emit(Op::Luminance { a, coeffs }) + } + "dotproduct" => { + let a = self.input_or(node, "in1", Val::ZERO); + let b = self.input_or(node, "in2", Val::ZERO); + self.emit(Op::DotProduct { a, b }) + } + "mix" => { + let fg = self.input_or(node, "fg", Val::ZERO); + let bg = self.input_or(node, "bg", Val::ZERO); + let m = self.input_or(node, "mix", Val::ZERO); + self.emit(Op::Mix { fg, bg, m }) + } + "clamp" => { + let a = self.input_or(node, "in", Val::ZERO); + let low = self.input_or(node, "low", Val::ZERO); + let high = self.input_or(node, "high", Val::ONE); + self.emit(Op::Clamp { a, low, high }) + } + "contrast" => { + let a = self.input_or(node, "in", Val::ZERO); + let amount = self.input_or(node, "amount", Val::ONE); + let pivot = self.input_or(node, "pivot", Val::float(0.5)); + self.emit(Op::Contrast { a, amount, pivot }) + } + "remap" => { + let a = self.input_or(node, "in", Val::ZERO); + let in_low = self.input_or(node, "inlow", Val::ZERO); + let in_high = self.input_or(node, "inhigh", Val::ONE); + let out_low = self.input_or(node, "outlow", Val::ZERO); + let out_high = self.input_or(node, "outhigh", Val::ONE); + self.emit(Op::Remap { + a, + in_low, + in_high, + out_low, + out_high, + }) + } + "invert" => { + let a = self.input_or(node, "in", Val::ZERO); + let amount = self.input_or(node, "amount", Val::ONE); + self.emit(Op::Invert { a, amount }) + } + "smoothstep" => { + let a = self.input_or(node, "in", Val::ZERO); + let low = self.input_or(node, "low", Val::ZERO); + let high = self.input_or(node, "high", Val::ONE); + self.emit(Op::Smoothstep { a, low, high }) + } + "convert" => { + let a = self.input_or(node, "in", Val::ZERO); + self.emit(Op::Convert { a, arity }) + } + "extract" => { + let a = self.input_or(node, "in", Val::ZERO); + // `index` is an integer literal, so it is read off the raw + // text rather than through the float lanes. + let index = node + .input("index") + .and_then(|i| i.text.as_deref()) + .and_then(|t| t.trim().parse::().ok()) + .unwrap_or(0); + self.emit(Op::Extract { a, index }) + } + "combine2" => { + // The signature fixes each operand's width — `(float, float)` + // → `vector2`, `(color3, float)` → `color4`, and for a + // `vector4` `(vector3, float)` or `(vector2, vector2)`, told + // apart by either input's width (see `Compiler::input_arity`). + // Each operand is converted to its width first, so the + // concatenation is exact whatever produced it (an unauthored + // `in1` is a zero, which must still fill three lanes of a + // `color4`). + let declared = |n: &str| node.input(n).map(|i| self.input_arity(node, i)); + let (na, nb) = match arity { + 4 if declared("in1") == Some(2) || declared("in2") == Some(2) => (2, 2), + 4 => (3, 1), + _ => (1, 1), + }; + let a = self.input_or(node, "in1", Val::ZERO); + let a = self.emit(Op::Convert { a, arity: na }); + let b = self.input_or(node, "in2", Val::ZERO); + let b = self.emit(Op::Convert { a: b, arity: nb }); + self.emit(Op::Combine2 { a, b }) + } + "combine3" => { + let a = self.input_or(node, "in1", Val::ZERO); + let b = self.input_or(node, "in2", Val::ZERO); + let c = self.input_or(node, "in3", Val::ZERO); + self.emit(Op::Combine3 { a, b, c }) + } + "normalmap" => { + let a = self.input_or(node, "in", Val::vec3(0.5, 0.5, 1.0)); + let scale = self.input_or(node, "scale", Val::ONE); + self.emit(Op::NormalMap { a, scale }) + } + "artistic_ior" => { + let reflectivity = + self.input_or(node, "reflectivity", Val::vec3(0.944, 0.776, 0.373)); + let edge = self.input_or(node, "edge_color", Val::vec3(0.998, 0.981, 0.751)); + self.emit(Op::ArtisticIor { + reflectivity, + edge, + extinction: output == Some("extinction"), + }) + } + "chiang_hair_roughness" => self.compile_chiang_hair_roughness(node, output), + "chiang_hair_absorption_from_color" => { + self.compile_chiang_hair_absorption_from_color(node) + } + "deon_hair_absorption_from_melanin" => { + self.compile_deon_hair_absorption_from_melanin(node) + } + "colorcorrect" => self.compile_colorcorrect(node), + "heighttonormal" => { + let scale = self.input_or(node, "scale", Val::ONE); + // Half a footprint either side, so the difference spans one. + let xp = self.shifted_input(node, "in", [0.5, 0.0]); + let xm = self.shifted_input(node, "in", [-0.5, 0.0]); + let yp = self.shifted_input(node, "in", [0.0, 0.5]); + let ym = self.shifted_input(node, "in", [0.0, -0.5]); + self.emit(Op::HeightToNormal { + xp, + xm, + yp, + ym, + scale, + }) + } + other => { + self.unsupported.insert(other.to_string()); + // Degrade this input to mid-grey rather than to black: an + // unsupported *pattern* node is usually a colour correction, + // and zero would turn whatever it feeds into a hole. + self.constant(Val::float(0.5)) + } + } + } + + /// `input` of `node` compiled with every lookup under it displaced by a + /// further `shift` footprint widths. + fn shifted_input(&mut self, node: &Node, input: &str, shift: [f32; 2]) -> u32 { + let outer = self.uv_shift; + self.uv_shift = [outer[0] + shift[0], outer[1] + shift[1]]; + let slot = self.input_or(node, input, Val::ZERO); + self.uv_shift = outer; + slot + } + + /// MaterialX `colorcorrect` (`color3`), lowered to the stdlib's own + /// `NG_colorcorrect_color3` chain: `hsvadjust` (hue) → `saturate` → + /// `range` (gamma) → lift → gain → `contrast` → exposure. + /// + /// Expanded here rather than kept as one op so the stages reuse + /// operators the JIT already inlines. A stage whose parameter folds to + /// its identity (hue 0, saturation 1, …) is left out: the reference's + /// arithmetic at those values is the identity up to rounding at most, and + /// the playground's graphs author one or two of the eight inputs. + fn compile_colorcorrect(&mut self, node: &Node) -> u32 { + if node.type_name != "color3" { + // `color4` routes alpha around the chain, and there is no + // `combine4` to put it back with; say so rather than correct the + // alpha too. + self.unsupported + .insert(format!("colorcorrect ({})", node.type_name)); + return self.constant(Val::float(0.5)); + } + let input = self.input_or(node, "in", Val::vec3(1.0, 1.0, 1.0)); + // Promote a `float` input to the colour MaterialX makes of it — its + // first lane, three times — before any stage runs. A one-lane texture + // carries its file's other channels in the lanes above the first, and + // the stages below work lane by lane, so without this they would + // correct those hidden channels too and a later `convert` would + // surface them as colour. `zip` broadcasts a one-lane operand from + // lane 0 and multiplying by one is exact, so a `color3` input passes + // through bit for bit. + let ones = self.constant(Val::vec3(1.0, 1.0, 1.0)); + let mut c = self.emit(Op::Binary { + op: BinOp::Mul, + a: input, + b: ones, + }); + // Each parameter's slot, unless it folds to the stage's identity. + let param = |cc: &mut Self, name: &str, default: f32| -> Option { + let s = cc.input_or(node, name, Val::float(default)); + (cc.fold(s).map(|v| v.x()) != Some(default)).then_some(s) + }; + if let Some(hue) = param(self, "hue", 0.0) { + let one = self.constant(Val::ONE); + let amount = self.emit(Op::Combine3 { + a: hue, + b: one, + c: one, + }); + c = self.emit(Op::HsvAdjust { a: c, amount }); + } + if let Some(sat) = param(self, "saturation", 1.0) { + // `saturate`: mix from the luminance grey toward the colour. + let grey = self.luminance(c); + c = self.emit(Op::Mix { + fg: c, + bg: grey, + m: sat, + }); + } + if let Some(gamma) = param(self, "gamma", 1.0) { + // `range` over [0, 1] → [0, 1] unclamped: both remaps are the + // identity, leaving `sign(x)·|x|^(1/gamma)`. + let one = self.constant(Val::ONE); + let recip = self.emit(Op::Binary { + op: BinOp::Div, + a: one, + b: gamma, + }); + let abs = self.emit(Op::Unary { + op: UnOp::Abs, + a: c, + }); + let pow = self.emit(Op::Binary { + op: BinOp::Pow, + a: abs, + b: recip, + }); + let sign = self.emit(Op::Unary { + op: UnOp::Sign, + a: c, + }); + c = self.emit(Op::Binary { + op: BinOp::Mul, + a: pow, + b: sign, + }); + } + if let Some(lift) = param(self, "lift", 0.0) { + // `c·(1 − lift) + lift`: raises black to `lift`, keeps white. + let one = self.constant(Val::ONE); + let keep = self.emit(Op::Binary { + op: BinOp::Sub, + a: one, + b: lift, + }); + let scaled = self.emit(Op::Binary { + op: BinOp::Mul, + a: c, + b: keep, + }); + c = self.emit(Op::Binary { + op: BinOp::Add, + a: scaled, + b: lift, + }); + } + if let Some(gain) = param(self, "gain", 1.0) { + c = self.emit(Op::Binary { + op: BinOp::Mul, + a: c, + b: gain, + }); + } + if let Some(amount) = param(self, "contrast", 1.0) { + let pivot = self.input_or(node, "contrastpivot", Val::float(0.5)); + c = self.emit(Op::Contrast { + a: c, + amount, + pivot, + }); + } + if let Some(exposure) = param(self, "exposure", 0.0) { + let two = self.constant(Val::float(2.0)); + let k = self.emit(Op::Binary { + op: BinOp::Pow, + a: two, + b: exposure, + }); + c = self.emit(Op::Binary { + op: BinOp::Mul, + a: c, + b: k, + }); + } + c + } + + fn compile_image(&mut self, node: &Node, arity: u8) -> u32 { + let file = node.input("file"); + // The texels' colour space: the `file` input's effective one, for a + // colour image only. A `float` or `vector3` image is data — a mask, + // a height, a normal map — whatever the document's default says. + let space = file + .filter(|_| is_color_type(&node.type_name)) + .and_then(|f| self.doc.colorspace_of(node, f)) + .map(str::to_string); + let tex = file.and_then(|i| i.text.clone()).and_then(|f| { + let loader = self.host.load_texture; + self.images + .entry((f, space)) + .or_insert_with_key(|(f, s)| loader(f, s.as_deref())) + .clone() + }); + // The authored `default` is a literal like any other: a colour one is + // converted from its own effective space. + let fallback = self.literal_of(node, "default").unwrap_or(Val::float(0.5)); + // `tiledimage` scales and offsets the chart before the lookup; + // `image` samples it as authored. + let (scale, offset) = if node.category == "tiledimage" { + let s = self.static_input(node, "uvtiling", Val::vec2(1.0, 1.0)); + let o = self.static_input(node, "uvoffset", Val::vec2(0.0, 0.0)); + let s = if s.arity == 1 { + [s.x(), s.x()] + } else { + [s.v[0], s.v[1]] + }; + let o = if o.arity == 1 { + [o.x(), o.x()] + } else { + [o.v[0], o.v[1]] + }; + (s, o) + } else { + ([1.0, 1.0], [0.0, 0.0]) + }; + let coord = self.image_coord(node); + self.emit(Op::Texture { + tex, + fallback, + scale, + offset, + arity, + shift: self.uv_shift, + coord, + }) + } + + /// An image's authored `texcoord`, compiled, or `None` for the shading + /// point's own chart. + /// + /// `None` covers the unconnected input and a connection to the default + /// chart itself (`texcoord` index 0, or `geompropvalue` of `st`), which is + /// how nearly every document spells it; those keep the op on the JIT's + /// inline texture path. A connection whose subgraph meets an operator + /// this compiler lacks (`place2d`, a second UV set) also takes `None`: + /// the chart is a better stand-in for an unknown coordinate than the + /// constant the unknown node would compile to, and that node is already + /// reported. + fn image_coord(&mut self, node: &Node) -> Option { + let input = node.input("texcoord")?; + if let Source::Node { name, .. } = &input.source { + let scope = node.graph.clone().unwrap_or_default(); + if self.doc.find(&scope, name).is_some_and(is_default_chart) { + return None; + } + } + // A literal compiles to a constant: one fixed texel, as authored. + let reported = self.unsupported.len(); + let slot = self.compile_input(node, input); + (self.unsupported.len() == reported).then_some(slot) + } +} + +impl Compiler<'_> { + /// An authored literal `v` of `input` as the program holds it: a + /// `color3` / `color4` with an effective colour space converted into the + /// working space by the host, anything else as authored. + fn managed(&self, node: &Node, input: &Input, v: Val) -> Val { + if !is_color_type(&input.type_name) { + return v; + } + match self.doc.colorspace_of(node, input) { + Some(space) => convert_color(v, &input.type_name, |rgb| { + (self.host.convert_color)(space, rgb) + }), + None => v, + } + } + + /// `node`'s input `name` when it is an authored literal (not a + /// connection), with the value the program holds for it — converted, for + /// a colour. What literal-driven pruning reads, so that it tests exactly + /// the value every shading point would have computed. + pub(crate) fn literal_of(&self, node: &Node, name: &str) -> Option { + let input = node.input(name)?; + match input.source { + Source::Value(v) => Some(self.managed(node, input, v)), + _ => None, + } + } + + /// An input the lookup needs as a compile-time constant: the literal, or + /// a connection that folds to one (`convert(8.0)` is how documents feed a + /// `tiledimage`'s `uvtiling`). One that varies over the surface cannot be + /// baked into the texture op, so it takes `default` and is reported — + /// never silently. + fn static_input(&mut self, node: &Node, name: &str, default: Val) -> Val { + if let Some(v) = self.literal_of(node, name) { + return v; + } + if node.input(name).is_none() { + return default; + } + let slot = self.input_or(node, name, default); + self.fold(slot).unwrap_or_else(|| { + self.unsupported + .insert(format!("{} (varying {name})", node.category)); + default + }) + } +} + +/// Whether `node` is the shading point's default chart: `texcoord` with +/// index 0, or `geompropvalue` reading `st` — what `ShadeCtx::uv` holds. +fn is_default_chart(node: &Node) -> bool { + let text = |name: &str| { + node.input(name) + .and_then(|i| i.text.as_deref()) + .map(str::trim) + }; + match node.category.as_str() { + "texcoord" => text("index").is_none_or(|i| i == "0"), + "geompropvalue" => text("geomprop") == Some("st"), + _ => false, + } +} diff --git a/crates/crust-mtlx/src/eval/mod.rs b/crates/crust-mtlx/src/eval/mod.rs new file mode 100644 index 00000000..70462819 --- /dev/null +++ b/crates/crust-mtlx/src/eval/mod.rs @@ -0,0 +1,483 @@ +//! The MaterialX pattern graph, compiled to a flat program and run per hit. +//! +//! A `.mtlx` look-dev graph is evaluated *per shading point* — its textures, +//! masks and blends are the whole point — so this cannot be folded to +//! constants at import. But it must also not be walked as a name-addressed DOM +//! inside the integrator: every lookup would be a string hash, and the teapot's +//! ceramic graph alone is ~50 nodes consulted several times per path vertex +//! (once to sample, again for each NEE and guide evaluation). +//! +//! So the graph is **compiled once** into a [`Program`]: a topologically +//! ordered `Vec` whose operands are slot *indices* into the values +//! computed so far. Evaluation is then a linear scan with no branching on +//! names, no allocation (the value stack is a thread-local scratch buffer +//! reused across calls), and no hash lookups. +//! +//! The compiler is also where cycles and unknown operators are dealt with: +//! a node that cannot be compiled becomes a constant, so an unsupported +//! MaterialX node degrades that one input to its default rather than failing +//! the material. + +mod apply; +mod compile; +#[cfg(test)] +mod tests; + +use crate::texture::TextureRef; +use crate::value::Val; +use glam::Vec3A; + +use apply::apply; +pub use apply::{perturb_normal, reflectivity_from_ior}; +pub use compile::Compiler; + +/// The shading point a [`Program`] is evaluated at. +#[derive(Clone, Copy)] +pub struct ShadeCtx { + /// `primvars:st`, unwrapped — the integer part selects a UDIM tile. + pub uv: (f32, f32), + /// Geometric (ray-facing) world-space normal. + pub normal: Vec3A, + /// World-space tangent along increasing `u`; `ZERO` when unknown, which + /// makes `normalmap` pass the geometric normal straight through rather + /// than build a frame out of noise. + pub tangent: Vec3A, + /// Direction from the viewer *to* the surface — MaterialX's + /// `viewdirection`, which is the incoming ray's direction, not its + /// negation. + pub view: Vec3A, + pub position: Vec3A, + /// Diameter of the shading point's texture footprint, in the same UV + /// units as `uv` — what [`crate::Texture::eval`] filters over. `0.0` asks + /// every texture in the graph to point-sample, which is what a host that + /// tracks no footprint should leave it at. + pub uv_width: f32, +} + +/// A componentwise binary operator. +#[derive(Clone, Copy, Debug)] +pub enum BinOp { + Add, + Sub, + Mul, + Div, + Pow, + Min, + Max, + /// `a` with `b` lanes, for `combine`-style plumbing. + Modulo, +} + +/// A componentwise unary operator. +#[derive(Clone, Copy, Debug)] +pub enum UnOp { + Abs, + Ln, + Exp, + Sin, + Cos, + Asin, + Acos, + Sqrt, + Sign, + Floor, + Ceil, + Normalize, +} + +/// One instruction. Operands are slot indices, always strictly less than the +/// instruction's own slot — the compiler emits in topological order, so a +/// single forward pass evaluates the whole program. +#[derive(Clone, Debug)] +pub enum Op { + Const(Val), + /// A UV texture lookup. `tex` is `None` when the host declined the file, + /// in which case `fallback` — the node's `default` input, or mid-grey — + /// stands in, which is what keeps an undecodable texture from blackening + /// a surface. + Texture { + tex: Option, + fallback: Val, + /// `uvtiling` / `uvoffset` from a `tiledimage`; identity for `image`. + scale: [f32; 2], + offset: [f32; 2], + arity: u8, + /// Where to look up relative to the shading point, in footprint + /// widths (see `shifted_uv`). Zero everywhere except the copies of + /// a subgraph `heighttonormal` differentiates. + shift: [f32; 2], + /// The slot holding an authored `texcoord` connection, when the + /// image has one other than the default chart; `None` reads the + /// shading point's `uv` (displaced by `shift`). A connected + /// coordinate carries any shift in its own `TexCoord` ops, so a + /// coordinate that does not depend on the chart is not displaced. + coord: Option, + }, + /// `primvars:st` as a `vector2`, displaced by `shift` footprint widths + /// exactly as [`Op::Texture`] is. + TexCoord { + shift: [f32; 2], + }, + /// World-space geometric normal. + Normal, + ViewDirection, + Position, + Unary { + op: UnOp, + a: u32, + }, + Binary { + op: BinOp, + a: u32, + b: u32, + }, + /// `bg·(1−m) + fg·m`, MaterialX's `mix`. + Mix { + fg: u32, + bg: u32, + m: u32, + }, + Clamp { + a: u32, + low: u32, + high: u32, + }, + /// `(in − pivot)·amount + pivot`. + Contrast { + a: u32, + amount: u32, + pivot: u32, + }, + /// Linear rescale from `[inlow, inhigh]` to `[outlow, outhigh]`. + Remap { + a: u32, + in_low: u32, + in_high: u32, + out_low: u32, + out_high: u32, + }, + /// `amount − in`. + Invert { + a: u32, + amount: u32, + }, + /// Reinterpret at a different lane count, broadcasting a scalar. + Convert { + a: u32, + arity: u8, + }, + /// One lane of a wider value. + Extract { + a: u32, + index: usize, + }, + Combine3 { + a: u32, + b: u32, + c: u32, + }, + Combine2 { + a: u32, + b: u32, + }, + DotProduct { + a: u32, + b: u32, + }, + /// `dot(in.rgb, coeffs)`, keeping a `color4`'s alpha. + Luminance { + a: u32, + coeffs: u32, + }, + /// Decodes a tangent-space normal map into a world-space normal. + NormalMap { + a: u32, + scale: u32, + }, + /// Gulbrandsen's artist-friendly metal parameterisation. `extinction` + /// selects which of the node's two outputs this slot holds. + ArtisticIor { + reflectivity: u32, + edge: u32, + extinction: bool, + }, + /// Hermite interpolation between two edges. + Smoothstep { + a: u32, + low: u32, + high: u32, + }, + /// MaterialX `hsvadjust`: to HSV, add `amount.x` to the hue and scale + /// saturation and value by `amount.y` / `amount.z`, and back. + HsvAdjust { + a: u32, + amount: u32, + }, + /// MaterialX `heighttonormal`, from the height at half a footprint either + /// side of the shading point along `u` (`xp`, `xm`) and `v` (`yp`, `ym`). + /// The result is encoded in `[0, 1]`, like a normal-map texel. + HeightToNormal { + xp: u32, + xm: u32, + yp: u32, + ym: u32, + scale: u32, + }, +} + +impl Op { + /// Calls `f` on every operand slot index, in a fixed order. + pub fn for_each_operand(&mut self, mut f: impl FnMut(&mut u32)) { + match self { + Op::Texture { coord, .. } => { + if let Some(c) = coord { + f(c); + } + } + Op::Const(_) | Op::TexCoord { .. } | Op::Normal | Op::ViewDirection | Op::Position => {} + Op::Unary { a, .. } | Op::Convert { a, .. } | Op::Extract { a, .. } => f(a), + Op::Binary { a, b, .. } + | Op::Invert { a, amount: b } + | Op::Combine2 { a, b } + | Op::Luminance { a, coeffs: b } + | Op::DotProduct { a, b } + | Op::NormalMap { a, scale: b } + | Op::HsvAdjust { a, amount: b } + | Op::ArtisticIor { + reflectivity: a, + edge: b, + .. + } => { + f(a); + f(b); + } + Op::Mix { fg, bg, m } => { + f(fg); + f(bg); + f(m); + } + Op::Clamp { a, low, high } | Op::Smoothstep { a, low, high } => { + f(a); + f(low); + f(high); + } + Op::Contrast { a, amount, pivot } => { + f(a); + f(amount); + f(pivot); + } + Op::Combine3 { a, b, c } => { + f(a); + f(b); + f(c); + } + Op::Remap { + a, + in_low, + in_high, + out_low, + out_high, + } => { + f(a); + f(in_low); + f(in_high); + f(out_low); + f(out_high); + } + Op::HeightToNormal { + xp, + xm, + yp, + ym, + scale, + } => { + f(xp); + f(xm); + f(yp); + f(ym); + f(scale); + } + } + } + + /// Whether the result depends on nothing but the operands — no shading + /// point, no texture. Such an op over constant operands is a constant. + fn is_pure(&self) -> bool { + match self { + Op::Texture { tex, .. } => tex.is_none(), + Op::TexCoord { .. } + | Op::Normal + | Op::ViewDirection + | Op::Position + | Op::NormalMap { .. } => false, + _ => true, + } + } +} + +/// A compiled pattern graph. +/// +/// Slots `0..consts.len()` hold `consts`, copied in once per evaluation; slot +/// `consts.len() + i` holds the value of `ops[i]`. The compiler leaves +/// `consts` empty and emits every literal as an [`Op::Const`]; +/// [`Program::optimize`] moves them (and everything computable from them) +/// into `consts`. +#[derive(Clone, Default)] +pub struct Program { + pub consts: Vec, + pub ops: Vec, +} + +impl Program { + /// Number of slots an evaluation fills. + pub fn len(&self) -> usize { + self.consts.len() + self.ops.len() + } + + pub fn is_empty(&self) -> bool { + self.len() == 0 + } + + /// Whether every operand refers to a slot strictly before its user's — + /// what the compiler always emits, and what anything that runs a program + /// by other means than [`Program::eval`]'s bounds-checked reads relies on. + pub fn is_well_formed(&self) -> bool { + let nc = self.consts.len(); + self.ops.iter().enumerate().all(|(i, op)| { + let mut ok = true; + op.clone() + .for_each_operand(|o| ok &= (*o as usize) < nc + i); + ok + }) + } + + /// One instruction's value, given the slots computed before it — the + /// interpreter's own step, for an evaluator that runs some ops another + /// way and hands the rest back here (crust-jit), so that both produce the + /// same bits by construction. + pub fn apply_op(op: &Op, slots: &[Val], ctx: &ShadeCtx) -> Val { + apply(op, slots, ctx) + } + + /// Evaluates every instruction into `slots`, which is resized as needed. + /// + /// The caller owns the buffer so it can be reused across shading calls — + /// a fresh `Vec` per call would allocate once per BSDF evaluation, which + /// at several evaluations per path vertex is the kind of cost that does + /// not show up in a profile as one hot line. + pub fn eval(&self, ctx: &ShadeCtx, slots: &mut Vec) { + slots.clear(); + slots.reserve(self.len()); + slots.extend_from_slice(&self.consts); + for op in &self.ops { + let v = apply(op, slots, ctx); + slots.push(v); + } + } + + /// Rewrites the program so it computes the same values at `roots` with + /// less work per evaluation, and returns where each old slot went + /// (`None` for a slot that no longer exists). + /// + /// Three passes, each exact rather than approximate: + /// + /// - **Constant folding.** An op whose operands are all constant, and + /// which reads neither the shading point nor a texture, is evaluated + /// here — by `apply`, the function the interpreter runs, so the value + /// is the one every hit would have computed, bit for bit. + /// - **Constant hoisting and deduplication.** Constants move into + /// [`Program::consts`], copied in with one `memcpy` per evaluation + /// instead of one dispatched instruction each, and bitwise-equal ones + /// share a slot (the compiler emits a fresh constant for every literal + /// and every unauthored input's default). + /// - **Dead-code elimination.** Ops nothing at `roots` depends on are + /// dropped. + /// + /// The surviving ops keep their relative order, so operands still precede + /// their users. + /// + /// A malformed program — an operand that does not precede its user, or a + /// root out of range — is returned unchanged with the identity remap: the + /// interpreter already reads such an operand as zero, and rewriting it + /// would have to invent a meaning for it. + pub fn optimize(&self, roots: &[u32]) -> (Program, Vec>) { + let n = self.len(); + let nc = self.consts.len(); + if !self.is_well_formed() || roots.iter().any(|&r| r as usize >= n) { + return (self.clone(), (0..n as u32).map(Some).collect()); + } + // Every slot's value, where it is a compile-time constant. + let mut known: Vec> = self.consts.iter().copied().map(Some).collect(); + known.resize(n, None); + let ctx = ShadeCtx { + uv: (0.0, 0.0), + normal: Vec3A::Z, + tangent: Vec3A::ZERO, + view: -Vec3A::Z, + position: Vec3A::ZERO, + uv_width: 0.0, + }; + let mut scratch: Vec = vec![Val::ZERO; n]; + for (i, op) in self.ops.iter().enumerate() { + let slot = nc + i; + let mut all_known = op.is_pure(); + op.clone() + .for_each_operand(|o| match known.get(*o as usize).copied().flatten() { + Some(v) => scratch[*o as usize] = v, + None => all_known = false, + }); + if all_known { + // `apply` reads operands by slot, so hand it the known values + // at their own indices. + known[slot] = Some(apply(op, &scratch[..slot], &ctx)); + } + } + + // Liveness, from the roots back. + let mut live = vec![false; n]; + for &r in roots { + if let Some(l) = live.get_mut(r as usize) { + *l = true; + } + } + for i in (0..self.ops.len()).rev() { + let slot = nc + i; + if live[slot] && known[slot].is_none() { + self.ops[i].clone().for_each_operand(|o| { + if let Some(l) = live.get_mut(*o as usize) { + *l = true; + } + }); + } + } + + // Constants first, deduplicated bitwise, then the surviving ops. + let mut out = Program::default(); + let mut remap: Vec> = vec![None; n]; + let key = |v: &Val| (v.v.map(f32::to_bits), v.arity); + let mut seen: std::collections::HashMap<([u32; 4], u8), u32> = Default::default(); + for slot in 0..n { + if let (true, Some(v)) = (live[slot], known[slot]) { + let idx = *seen.entry(key(&v)).or_insert_with(|| { + out.consts.push(v); + (out.consts.len() - 1) as u32 + }); + remap[slot] = Some(idx); + } + } + let base = out.consts.len(); + for (i, op) in self.ops.iter().enumerate() { + let slot = nc + i; + if live[slot] && known[slot].is_none() { + let mut op = op.clone(); + op.for_each_operand(|o| { + // A live op's operands are live, so they were placed. + *o = remap[*o as usize].unwrap_or(0); + }); + remap[slot] = Some((base + out.ops.len()) as u32); + out.ops.push(op); + } + } + (out, remap) + } +} diff --git a/crates/crust-mtlx/src/eval/tests.rs b/crates/crust-mtlx/src/eval/tests.rs new file mode 100644 index 00000000..b928f152 --- /dev/null +++ b/crates/crust-mtlx/src/eval/tests.rs @@ -0,0 +1,525 @@ +use glam::Vec3A; + +use super::apply::*; +use super::*; +use crate::Host; +use crate::parse::Doc; +use crate::texture::TextureRef; +use crate::value::Val; + +fn no_textures(_: &str, _: Option<&str>) -> Option { + None +} + +fn run(doc_text: &str, node: &str) -> Val { + let doc = Doc::parse(doc_text).unwrap(); + let loader = no_textures; + let mut c = Compiler::new(&doc, &Host::new(&loader)); + let slot = c.compile_named("", node, None); + let mut slots = Vec::new(); + c.program.eval( + &ShadeCtx { + uv: (0.25, 0.75), + normal: Vec3A::Z, + tangent: Vec3A::X, + view: -Vec3A::Z, + position: Vec3A::ZERO, + uv_width: 0.0, + }, + &mut slots, + ); + slots[slot as usize] +} + +#[test] +fn mix_blends_bg_toward_fg() { + let v = run( + r#" + + + + + + + + "#, + "m", + ); + assert!((v.x() - 0.25).abs() < 1e-6, "got {}", v.x()); +} + +#[test] +fn artistic_ior_round_trips_to_its_reflectivity() { + // The conductor lobe reduces (n, k) back to a reflectivity colour, so + // the two conversions must be inverses or every metal shifts hue. + for r in [0.05f32, 0.4, 0.94] { + let (n, k) = artistic_ior(Vec3A::splat(r), Vec3A::ONE); + let back = reflectivity_from_ior(n, k); + assert!((back.x - r).abs() < 1e-4, "r={r} -> {}", back.x); + } +} + +#[test] +fn a_graph_output_keeps_the_output_it_selected() { + // A ``'s `` may select one output of a multioutput + // node. Dropping that selection is silent: the reference falls back to + // the node's first output, so a graph publishing `artistic_ior`'s + // extinction hands its consumer the ior instead. + let doc = r#" + + + + + + + + + + + + + + + + + "#; + let (n, k) = artistic_ior(Vec3A::splat(0.5), Vec3A::ONE); + let got_n = run(doc, "take_n"); + let got_k = run(doc, "take_k"); + assert!((got_n.x() - n.x).abs() < 1e-5, "ior: got {}", got_n.x()); + assert!( + (got_k.x() - k.x).abs() < 1e-5, + "extinction: got {}", + got_k.x() + ); + // The two outputs are what the test is about; equal values would make + // the assertions above pass for the wrong reason. + assert!((n.x - k.x).abs() > 1e-3); +} + +#[test] +fn a_graph_output_without_a_selection_takes_the_first_output() { + // The single-output majority authors no `output` attribute, and must + // keep resolving as it did. + let v = run( + r#" + + + + + + + + + + + + "#, + "take", + ); + let (n, _) = artistic_ior(Vec3A::splat(0.5), Vec3A::ONE); + assert!((v.x() - n.x).abs() < 1e-5, "got {}", v.x()); +} + +#[test] +fn a_cycle_terminates_instead_of_overflowing_the_stack() { + let v = run( + r#" + + + + + + + "#, + "a", + ); + assert!(v.x().is_finite()); +} + +/// `colorcorrect` of a constant `in`, with `params` authored as floats. +fn colorcorrect(input: [f32; 3], params: &[(&str, f32)]) -> Vec3A { + let inputs: String = params + .iter() + .map(|(n, v)| format!(r#""#)) + .collect(); + let doc = format!( + r#" + + + {inputs} + + "#, + input[0], input[1], input[2] + ); + let v = run(&doc, "cc"); + assert_eq!(v.arity, 3); + v.rgb() +} + +fn close(got: Vec3A, want: Vec3A) { + assert!( + (got - want).abs().max_element() < 1e-5, + "got {got}, want {want}" + ); +} + +#[test] +fn colorcorrect_defaults_are_the_identity() { + // Bitwise: every stage folds to its identity and is left out. + let c = [0.1, 0.7, 2.5]; + assert_eq!(colorcorrect(c, &[]), Vec3A::from(c)); + assert_eq!( + colorcorrect(c, &[("hue", 0.0), ("gain", 1.0), ("exposure", 0.0)]), + Vec3A::from(c) + ); +} + +#[test] +fn colorcorrect_applies_each_stage_as_the_stdlib_graph_does() { + let c = [0.125, 0.5, 1.0]; + close(colorcorrect(c, &[("gain", 4.0)]), Vec3A::new(0.5, 2.0, 4.0)); + // `range` with gamma: x^(1/gamma), sign-preserving, unclamped. + close( + colorcorrect([0.125, 8.0, -0.125], &[("gamma", 3.0)]), + Vec3A::new(0.5, 2.0, -0.5), + ); + // Lift raises black to `lift` and leaves white alone. + close( + colorcorrect([0.0, 0.5, 1.0], &[("lift", 0.5)]), + Vec3A::new(0.5, 0.75, 1.0), + ); + close( + colorcorrect( + [0.25, 0.5, 1.0], + &[("contrast", 2.0), ("contrastpivot", 0.5)], + ), + Vec3A::new(0.0, 0.5, 1.5), + ); + close( + colorcorrect(c, &[("exposure", -1.0)]), + Vec3A::new(0.0625, 0.25, 0.5), + ); + // Saturation 0 is the luminance grey (MaterialX's ACEScg default + // `lumacoeffs`); 2 pushes away from it. + let l = 0.2722287 * 0.125 + 0.6740818 * 0.5 + 0.0536895; + close(colorcorrect(c, &[("saturation", 0.0)]), Vec3A::splat(l)); + close( + colorcorrect(c, &[("saturation", 2.0)]), + Vec3A::from(c) * 2.0 - Vec3A::splat(l), + ); + // Hue rotates in turns and wraps: a third of a turn takes red to + // green, and 1.5 turns is the same as a half. + close( + colorcorrect([1.0, 0.0, 0.0], &[("hue", 1.0 / 3.0)]), + Vec3A::new(0.0, 1.0, 0.0), + ); + close( + colorcorrect([1.0, 0.0, 0.0], &[("hue", 1.5)]), + Vec3A::new(0.0, 1.0, 1.0), + ); +} + +#[test] +fn colorcorrect_stages_run_in_the_stdlib_order() { + // Gamma before lift before gain before contrast before exposure: any + // two swapped gives a different answer on this input. + let got = colorcorrect( + [0.25; 3], + &[ + ("gamma", 2.0), + ("lift", 0.5), + ("gain", 2.0), + ("contrast", 0.5), + ("exposure", 1.0), + ], + ); + let x = 0.25f32.sqrt(); // gamma → 0.5 + let x = x * (1.0 - 0.5) + 0.5; // lift → 0.75 + let x = x * 2.0; // gain → 1.5 + let x = (x - 0.5) * 0.5 + 0.5; // contrast → 1.0 + let x = x * 2.0; // exposure → 2.0 + close(got, Vec3A::splat(x)); +} + +#[test] +fn hsv_round_trips() { + for c in [ + Vec3A::new(0.9, 0.2, 0.1), + Vec3A::new(0.1, 0.8, 0.3), + Vec3A::new(0.2, 0.3, 0.95), + Vec3A::new(0.7, 0.1, 0.6), + Vec3A::splat(0.4), + Vec3A::new(3.0, 1.0, 0.5), + ] { + close(hsv_to_rgb(rgb_to_hsv(c)), c); + } +} + +/// A height of `slope_u · u + slope_v · v`, sampled through a real +/// `image` node so the shifted lookups are what is being differentiated. +struct Ramp { + slope_u: f32, + slope_v: f32, +} +impl crate::Texture for Ramp { + fn eval(&self, u: f32, v: f32, _: f32) -> [f32; 4] { + [self.slope_u * u + self.slope_v * v; 4] + } +} + +const HEIGHT_DOC: &str = r#" + + + + + + + + + + + "#; + +fn height_to_normal_at(slope_u: f32, slope_v: f32, uv_width: f32, node: &str) -> Vec3A { + let doc = Doc::parse(HEIGHT_DOC).unwrap(); + let loader = move |_: &str, _: Option<&str>| { + Some(TextureRef(std::sync::Arc::new(Ramp { slope_u, slope_v }))) + }; + let mut c = Compiler::new(&doc, &Host::new(&loader)); + let slot = c.compile_named("", node, None); + let mut slots = Vec::new(); + let ctx = ShadeCtx { + uv: (0.3, 0.6), + normal: Vec3A::Z, + tangent: Vec3A::X, + view: -Vec3A::Z, + position: Vec3A::ZERO, + uv_width, + }; + c.program.eval(&ctx, &mut slots); + slots[slot as usize].rgb() +} + +#[test] +fn heighttonormal_is_the_osl_reference_over_one_footprint() { + // A height rising 10 per UV unit, over a 0.01-wide footprint, changes + // by 0.1 across it: the reference's `Dx(in)`. + let (du, dv) = (0.1f32, 0.0f32); + let dz = (1.0 - du * du - dv * dv).sqrt(); + let want = Vec3A::new(-du, -dv, dz).normalize() * 0.5 + Vec3A::splat(0.5); + close(height_to_normal_at(10.0, 0.0, 0.01, "n"), want); + let want = Vec3A::new(0.0, -0.1, dz).normalize() * 0.5 + Vec3A::splat(0.5); + close(height_to_normal_at(0.0, 10.0, 0.01, "n"), want); +} + +#[test] +fn heighttonormal_tilts_away_from_rising_height() { + // Through `normalmap` in a frame with the tangent along +u: a height + // rising along +u (+v) is a slope facing −u (−v), which is where the + // normal must lean. + let n = height_to_normal_at(10.0, 0.0, 0.01, "world"); + assert!(n.x < -0.05 && n.y.abs() < 1e-6 && n.z > 0.9, "{n}"); + let n = height_to_normal_at(0.0, 10.0, 0.01, "world"); + assert!(n.y < -0.05 && n.x.abs() < 1e-6 && n.z > 0.9, "{n}"); +} + +#[test] +fn heighttonormal_without_a_footprint_is_flat() { + // No ray cone, no derivative — the nodedef's own default output. + assert_eq!( + height_to_normal_at(10.0, 5.0, 0.0, "n"), + Vec3A::new(0.5, 0.5, 1.0) + ); +} + +/// `node` of `doc`, every image in it served by `tex`, at uv (0.3, 0.6) +/// with a footprint `uv_width` wide. Also returns the compiled program. +fn eval_with( + doc: &str, + node: &str, + tex: impl crate::Texture + Clone + 'static, + uv_width: f32, +) -> (Val, Program) { + let doc = Doc::parse(doc).unwrap(); + let loader = move |_: &str, _: Option<&str>| Some(TextureRef(std::sync::Arc::new(tex.clone()))); + let mut c = Compiler::new(&doc, &Host::new(&loader)); + let slot = c.compile_named("", node, None); + let mut slots = Vec::new(); + let ctx = ShadeCtx { + uv: (0.3, 0.6), + normal: Vec3A::Z, + tangent: Vec3A::X, + view: -Vec3A::Z, + position: Vec3A::ZERO, + uv_width, + }; + c.program.eval(&ctx, &mut slots); + (slots[slot as usize], c.program) +} + +#[derive(Clone)] +struct RampTex(f32); +impl crate::Texture for RampTex { + fn eval(&self, u: f32, _: f32, _: f32) -> [f32; 4] { + [self.0 * u; 4] + } +} + +/// `heighttonormal` over an image whose `texcoord` is `coord` — a +/// fragment of nodes naming the connected one `c`, or empty for none. +fn height_doc(coord: &str) -> String { + let conn = if coord.is_empty() { + String::new() + } else { + r#""#.into() + }; + format!( + r#" + {coord} + + + {conn} + + + + + "# + ) +} + +fn texture_coords(p: &Program) -> Vec> { + p.ops + .iter() + .filter_map(|op| match op { + Op::Texture { coord, .. } => Some(*coord), + _ => None, + }) + .collect() +} + +#[test] +fn heighttonormal_of_a_constant_coordinate_is_flat() { + // Every tap reads the same texel, whatever the footprint. + let doc = height_doc( + r#""#, + ); + let (n, _) = eval_with(&doc, "n", RampTex(10.0), 0.01); + assert_eq!(n.rgb(), Vec3A::new(0.5, 0.5, 1.0)); +} + +#[test] +fn the_default_chart_spelled_out_is_the_implicit_one() { + // `geompropvalue st` and `texcoord` are what `ctx.uv` already holds: + // same answer as no connection, and still the JIT's inline lookup. + let (want, p) = eval_with(&height_doc(""), "n", RampTex(10.0), 0.01); + assert!(texture_coords(&p).iter().all(Option::is_none)); + for c in [ + r#""#, + r#""#, + ] { + let (got, p) = eval_with(&height_doc(c), "n", RampTex(10.0), 0.01); + assert_eq!(bits_of(got), bits_of(want), "{c}"); + assert!(texture_coords(&p).iter().all(Option::is_none), "{c}"); + } +} + +fn bits_of(v: Val) -> ([u32; 4], u8) { + (v.v.map(f32::to_bits), v.arity) +} + +#[test] +fn an_authored_coordinate_is_sampled_and_differentiated_through() { + // `texcoord · 2`: the image reads at twice the chart, and the height + // changes twice as fast across the same footprint. + let doc = height_doc( + r#" + + + + "#, + ); + let (h, _) = eval_with(&doc, "h", RampTex(10.0), 0.01); + assert!((h.x() - 10.0 * 0.6).abs() < 1e-5, "height {}", h.x()); + let (n, _) = eval_with(&doc, "n", RampTex(10.0), 0.01); + let du = 0.2f32; // 10 per unit, 2x the chart, 0.01 across + let want = Vec3A::new(-du, 0.0, (1.0 - du * du).sqrt()).normalize() * 0.5 + Vec3A::splat(0.5); + close(n.rgb(), want); +} + +#[test] +fn an_uncompilable_coordinate_falls_back_to_the_chart() { + // `place2d` has no operator here; its constant stand-in would pin the + // lookup to one texel, so the chart is used instead (and reported). + let doc = height_doc(r#""#); + let (h, p) = eval_with(&doc, "h", RampTex(10.0), 0.01); + assert!((h.x() - 3.0).abs() < 1e-5, "height {}", h.x()); + assert!(texture_coords(&p).iter().all(Option::is_none)); +} + +#[derive(Clone)] +struct Channels; +impl crate::Texture for Channels { + fn eval(&self, _: f32, _: f32, _: f32) -> [f32; 4] { + [0.25, 0.5, 0.75, 1.0] + } +} + +#[test] +fn colorcorrect_of_a_float_image_is_grey() { + // A one-lane lookup keeps its file's other channels in lanes 1..3; the + // correction must promote lane 0, not correct and expose the rest. + let doc = r#" + + + + + + + + + + + "#; + for node in ["cc", "out"] { + let (v, _) = eval_with(doc, node, Channels, 0.0); + assert_eq!(v.arity, 3, "{node}"); + assert_eq!(v.rgb(), Vec3A::splat(0.5), "{node}"); + } +} + +#[test] +fn heighttonormal_asks_the_host_for_its_image_once() { + // Four shifted copies of the lookup, one sampler: the host's decode + // and residency are per file, not per tap. + let doc = Doc::parse(HEIGHT_DOC).unwrap(); + let calls = std::cell::Cell::new(0); + let loader = |_: &str, _: Option<&str>| { + calls.set(calls.get() + 1); + Some(TextureRef(std::sync::Arc::new(Ramp { + slope_u: 1.0, + slope_v: 0.0, + }))) + }; + let mut c = Compiler::new(&doc, &Host::new(&loader)); + c.compile_named("", "n", None); + c.compile_named("", "h", None); + assert_eq!(calls.get(), 1); +} + +#[test] +fn division_by_zero_stays_finite() { + // `1 / transmittance` with a black channel is authored in the teapot's + // own ceramic graph; an infinity there survives every later multiply + // and reaches the framebuffer as a NaN pixel. + let v = run( + r#" + + + + + "#, + "d", + ); + assert!(v.x().is_finite()); +} diff --git a/crates/crust-mtlx/src/surface.rs b/crates/crust-mtlx/src/surface.rs deleted file mode 100644 index 378ed464..00000000 --- a/crates/crust-mtlx/src/surface.rs +++ /dev/null @@ -1,1362 +0,0 @@ -//! MaterialX surface-shader nodes expanded into the closure tree of their -//! implementation nodegraphs. -//! -//! `open_pbr_surface`, `standard_surface` and `gltf_pbr` are, in MaterialX, -//! nodedefs whose implementations are nodegraphs over standalone BSDF nodes -//! (`libraries/bxdf/*.mtlx`, MaterialX 1.39). A document authoring one never -//! carries the implementation, so this module reproduces each graph **node for -//! node** — the same leaves, the same `layer` / `mix` / `multiply` in the same -//! order, and the same derived parameters, which are emitted as program ops so -//! they fold, optimise and JIT like any pattern graph. Each block below is -//! commented with the nodegraph node names it reproduces, so it can be checked -//! against the `.mtlx` line by line. NVIDIA Typhoon builds its surfaces the -//! same way (`MaterialXCpp/materials/*.cpp`); where Typhoon departs from the -//! graph (it omits OpenPBR's thin-walled subsurface branch), the graph wins. -//! -//! Every input takes its connection, else its authored value, else its -//! nodedef default from the tables here — generated from the nodedefs, which -//! are vendored under `tests/nodedefs/` and checked against these tables. -//! -//! Two parts of each graph sit outside the closure tree and are carried beside -//! it: the `surface` node's `opacity` ([`Closures::opacity`]), and the -//! `rotate3d` a graph applies to its tangent, which becomes the angle a leaf's -//! frame is turned by ([`Leaf::rotation`]). -//! -//! What the tree cannot represent is reported rather than dropped silently: -//! glTF occlusion, and inputs the MaterialX graphs themselves ignore. - -use crate::bsdf::{ - Bsdf, Closure, Closures, DiffuseModel, EdfFalloff, Emission, Leaf, NodeId, ScatterMode, - SheenMode, Slot, ThinFilm, Volume, -}; -use crate::eval::{BinOp, Compiler, Op, UnOp}; -use crate::parse::Node; -use crate::value::Val; -use std::collections::HashMap; - -/// One nodedef input: its name, MaterialX type and default. `None` for an -/// input whose default is a geometric property (`Nworld`, `Tworld`) or that -/// declares no value. -#[derive(Clone, Copy, Debug)] -pub struct InputDef { - pub name: &'static str, - pub ty: &'static str, - pub default: Option, -} - -// Generated from `tests/nodedefs/*.mtlx` (MaterialX 1.39, Apache-2.0); the -// `nodedef_tables_match_materialx` test keeps them honest. `standard_surface` -// is version 1.0.1, which inherits 1.0.0 and overrides `base` and -// `base_color`. -#[rustfmt::skip] -mod tables { - use super::InputDef; - use crate::value::Val; - /// `ND_open_pbr_surface_surfaceshader`'s inputs, in nodedef order. - pub const OPEN_PBR_SURFACE: &[InputDef] = &[ - InputDef { name: "base_weight", ty: "float", default: Some(Val::float(1.0)) }, - InputDef { name: "base_color", ty: "color3", default: Some(Val::vec3(0.8, 0.8, 0.8)) }, - InputDef { name: "base_diffuse_roughness", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "base_metalness", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "specular_weight", ty: "float", default: Some(Val::float(1.0)) }, - InputDef { name: "specular_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, - InputDef { name: "specular_roughness", ty: "float", default: Some(Val::float(0.3)) }, - InputDef { name: "specular_ior", ty: "float", default: Some(Val::float(1.5)) }, - InputDef { name: "specular_roughness_anisotropy", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "transmission_weight", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "transmission_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, - InputDef { name: "transmission_depth", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "transmission_scatter", ty: "color3", default: Some(Val::vec3(0.0, 0.0, 0.0)) }, - InputDef { name: "transmission_scatter_anisotropy", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "transmission_dispersion_scale", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "transmission_dispersion_abbe_number", ty: "float", default: Some(Val::float(20.0)) }, - InputDef { name: "subsurface_weight", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "subsurface_color", ty: "color3", default: Some(Val::vec3(0.8, 0.8, 0.8)) }, - InputDef { name: "subsurface_radius", ty: "float", default: Some(Val::float(1.0)) }, - InputDef { name: "subsurface_radius_scale", ty: "color3", default: Some(Val::vec3(1.0, 0.5, 0.25)) }, - InputDef { name: "subsurface_scatter_anisotropy", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "fuzz_weight", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "fuzz_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, - InputDef { name: "fuzz_roughness", ty: "float", default: Some(Val::float(0.5)) }, - InputDef { name: "coat_weight", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "coat_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, - InputDef { name: "coat_roughness", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "coat_roughness_anisotropy", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "coat_ior", ty: "float", default: Some(Val::float(1.6)) }, - InputDef { name: "coat_darkening", ty: "float", default: Some(Val::float(1.0)) }, - InputDef { name: "thin_film_weight", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "thin_film_thickness", ty: "float", default: Some(Val::float(0.5)) }, - InputDef { name: "thin_film_ior", ty: "float", default: Some(Val::float(1.4)) }, - InputDef { name: "emission_luminance", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "emission_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, - InputDef { name: "geometry_opacity", ty: "float", default: Some(Val::float(1.0)) }, - InputDef { name: "geometry_thin_walled", ty: "boolean", default: Some(Val::float(0.0)) }, - InputDef { name: "geometry_normal", ty: "vector3", default: None }, - InputDef { name: "geometry_coat_normal", ty: "vector3", default: None }, - InputDef { name: "geometry_tangent", ty: "vector3", default: None }, - InputDef { name: "geometry_coat_tangent", ty: "vector3", default: None }, - ]; - - /// `ND_standard_surface_surfaceshader`'s inputs, in nodedef order. - pub const STANDARD_SURFACE: &[InputDef] = &[ - InputDef { name: "base", ty: "float", default: Some(Val::float(1.0)) }, - InputDef { name: "base_color", ty: "color3", default: Some(Val::vec3(0.8, 0.8, 0.8)) }, - InputDef { name: "diffuse_roughness", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "metalness", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "specular", ty: "float", default: Some(Val::float(1.0)) }, - InputDef { name: "specular_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, - InputDef { name: "specular_roughness", ty: "float", default: Some(Val::float(0.2)) }, - InputDef { name: "specular_IOR", ty: "float", default: Some(Val::float(1.5)) }, - InputDef { name: "specular_anisotropy", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "specular_rotation", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "transmission", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "transmission_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, - InputDef { name: "transmission_depth", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "transmission_scatter", ty: "color3", default: Some(Val::vec3(0.0, 0.0, 0.0)) }, - InputDef { name: "transmission_scatter_anisotropy", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "transmission_dispersion", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "transmission_extra_roughness", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "subsurface", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "subsurface_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, - InputDef { name: "subsurface_radius", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, - InputDef { name: "subsurface_scale", ty: "float", default: Some(Val::float(1.0)) }, - InputDef { name: "subsurface_anisotropy", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "sheen", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "sheen_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, - InputDef { name: "sheen_roughness", ty: "float", default: Some(Val::float(0.3)) }, - InputDef { name: "coat", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "coat_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, - InputDef { name: "coat_roughness", ty: "float", default: Some(Val::float(0.1)) }, - InputDef { name: "coat_anisotropy", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "coat_rotation", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "coat_IOR", ty: "float", default: Some(Val::float(1.5)) }, - InputDef { name: "coat_normal", ty: "vector3", default: None }, - InputDef { name: "coat_affect_color", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "coat_affect_roughness", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "thin_film_thickness", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "thin_film_IOR", ty: "float", default: Some(Val::float(1.5)) }, - InputDef { name: "emission", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "emission_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, - InputDef { name: "opacity", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, - InputDef { name: "thin_walled", ty: "boolean", default: Some(Val::float(0.0)) }, - InputDef { name: "normal", ty: "vector3", default: None }, - InputDef { name: "tangent", ty: "vector3", default: None }, - ]; - - /// `ND_gltf_pbr_surfaceshader`'s inputs, in nodedef order. - pub const GLTF_PBR: &[InputDef] = &[ - InputDef { name: "base_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, - InputDef { name: "metallic", ty: "float", default: Some(Val::float(1.0)) }, - InputDef { name: "roughness", ty: "float", default: Some(Val::float(1.0)) }, - InputDef { name: "normal", ty: "vector3", default: None }, - InputDef { name: "tangent", ty: "vector3", default: None }, - InputDef { name: "occlusion", ty: "float", default: Some(Val::float(1.0)) }, - InputDef { name: "transmission", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "specular", ty: "float", default: Some(Val::float(1.0)) }, - InputDef { name: "specular_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, - InputDef { name: "ior", ty: "float", default: Some(Val::float(1.5)) }, - InputDef { name: "alpha", ty: "float", default: Some(Val::float(1.0)) }, - InputDef { name: "alpha_mode", ty: "integer", default: Some(Val::float(0.0)) }, - InputDef { name: "alpha_cutoff", ty: "float", default: Some(Val::float(0.5)) }, - InputDef { name: "iridescence", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "iridescence_ior", ty: "float", default: Some(Val::float(1.3)) }, - InputDef { name: "iridescence_thickness", ty: "float", default: Some(Val::float(100.0)) }, - InputDef { name: "sheen_color", ty: "color3", default: Some(Val::vec3(0.0, 0.0, 0.0)) }, - InputDef { name: "sheen_roughness", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "clearcoat", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "clearcoat_roughness", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "clearcoat_normal", ty: "vector3", default: None }, - InputDef { name: "emissive", ty: "color3", default: Some(Val::vec3(0.0, 0.0, 0.0)) }, - InputDef { name: "emissive_strength", ty: "float", default: Some(Val::float(1.0)) }, - InputDef { name: "thickness", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "attenuation_distance", ty: "float", default: None }, - InputDef { name: "attenuation_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, - InputDef { name: "anisotropy_strength", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "anisotropy_rotation", ty: "float", default: Some(Val::float(0.0)) }, - InputDef { name: "dispersion", ty: "float", default: Some(Val::float(0.0)) }, - ]; - -} -pub use tables::{GLTF_PBR, OPEN_PBR_SURFACE, STANDARD_SURFACE}; - -/// Expands `node` — an `open_pbr_surface`, `standard_surface` or `gltf_pbr` — -/// into `out`. -pub fn build(c: &mut Compiler<'_>, node: &Node, out: &mut Closures) { - let (defs, default_version): (&'static [InputDef], &str) = match node.category.as_str() { - "open_pbr_surface" => (OPEN_PBR_SURFACE, "1.1"), - "standard_surface" => (STANDARD_SURFACE, "1.0.1"), - "gltf_pbr" => (GLTF_PBR, "2.0.1"), - other => { - c.unsupported.insert(other.to_string()); - return; - } - }; - let mut b = B { - c, - node, - defs, - memo: HashMap::new(), - out, - }; - if let Some(v) = b.version() - && v != default_version - { - b.out.reported.insert(format!( - "{} version {v} (built as the default version {default_version})", - node.category - )); - } - match node.category.as_str() { - "open_pbr_surface" => open_pbr_surface(&mut b), - "standard_surface" => standard_surface(&mut b), - _ => gltf_pbr(&mut b), - } -} - -/// A builder over one surface node: its inputs by name, program ops, and the -/// tree under construction. -struct B<'x, 'a> { - c: &'x mut Compiler<'a>, - node: &'x Node, - defs: &'static [InputDef], - memo: HashMap<&'static str, Slot>, - out: &'x mut Closures, -} - -impl B<'_, '_> { - fn version(&self) -> Option<&str> { - self.node.version.as_deref() - } - - fn def(&self, name: &str) -> Option<&'static InputDef> { - self.defs.iter().find(|d| d.name == name) - } - - /// An input: its connection, its authored value, or its nodedef default. - fn get(&mut self, name: &'static str) -> Slot { - if let Some(&s) = self.memo.get(name) { - return s; - } - let default = self.def(name).and_then(|d| d.default).unwrap_or(Val::ZERO); - let s = self.c.input_or(self.node, name, default); - self.memo.insert(name, s); - s - } - - /// An input whose default is geometric (`Nworld` / `Tworld`): `None` - /// unless authored. - fn geom(&mut self, name: &'static str) -> Option { - self.c.optional_input(self.node, name) - } - - /// Reports `name` when it is authored away from its nodedef default. - fn report(&mut self, name: &'static str, why: &str) { - let default = self.def(name).and_then(|d| d.default).unwrap_or(Val::ZERO); - if crate::bsdf::authored_away(self.c, self.node, name, default) { - self.out.reported.insert(format!("{name} ({why})")); - } - } - - fn k(&mut self, x: f32) -> Slot { - self.c.constant(Val::float(x)) - } - - fn k3(&mut self, x: f32, y: f32, z: f32) -> Slot { - self.c.constant(Val::vec3(x, y, z)) - } - - fn bin(&mut self, op: BinOp, a: Slot, b: Slot) -> Slot { - self.c.emit(Op::Binary { op, a, b }) - } - - fn mul(&mut self, a: Slot, b: Slot) -> Slot { - self.bin(BinOp::Mul, a, b) - } - - fn add(&mut self, a: Slot, b: Slot) -> Slot { - self.bin(BinOp::Add, a, b) - } - - fn sub(&mut self, a: Slot, b: Slot) -> Slot { - self.bin(BinOp::Sub, a, b) - } - - fn div(&mut self, a: Slot, b: Slot) -> Slot { - self.bin(BinOp::Div, a, b) - } - - fn un(&mut self, op: UnOp, a: Slot) -> Slot { - self.c.emit(Op::Unary { op, a }) - } - - fn mix(&mut self, fg: Slot, bg: Slot, m: Slot) -> Slot { - self.c.emit(Op::Mix { fg, bg, m }) - } - - fn clamp(&mut self, a: Slot, lo: f32, hi: f32) -> Slot { - let (low, high) = (self.k(lo), self.k(hi)); - self.c.emit(Op::Clamp { a, low, high }) - } - - fn convert(&mut self, a: Slot, arity: u8) -> Slot { - self.c.emit(Op::Convert { a, arity }) - } - - fn extract(&mut self, a: Slot, index: usize) -> Slot { - self.c.emit(Op::Extract { a, index }) - } - - /// MaterialX's `ifgreater`: `in1` where `value1 > value2`, else `in2`. - /// - /// Built from existing operators so the JIT needs nothing new: - /// `m = max(sign(value1 − value2), 0)` is 1 exactly when `value1` is - /// greater — MaterialX's `sign(0)` is 0, and `x − x` is `+0`, which is - /// what makes equality pick `in2` — and `mix(fg = in1, bg = in2, m)` is - /// exact at `m ∈ {0, 1}` for finite operands (every divide and log in the - /// program is kept finite). - fn gt(&mut self, v1: Slot, v2: Slot, in1: Slot, in2: Slot) -> Slot { - let d = self.sub(v1, v2); - let s = self.un(UnOp::Sign, d); - let zero = self.k(0.0); - let m = self.bin(BinOp::Max, s, zero); - self.mix(in1, in2, m) - } - - /// MaterialX's `ifgreatereq`: `in1` where `value1 ≥ value2`, else `in2` — - /// [`B::gt`] with its operands swapped, which is exact for the same reason. - fn ge(&mut self, v1: Slot, v2: Slot, in1: Slot, in2: Slot) -> Slot { - self.gt(v2, v1, in2, in1) - } - - /// MaterialX's `ifequal`: `in1` where `value1 = value2`, else `in2`, - /// built as "neither is greater". - fn eq(&mut self, v1: Slot, v2: Slot, in1: Slot, in2: Slot) -> Slot { - let below = self.gt(v2, v1, in2, in1); - self.gt(v1, v2, in2, below) - } - - /// The right-handed angle, in radians, a graph's `rotate3d` of its tangent - /// by `degrees` about the normal turns it by, `None` when `degrees` folds - /// to 0. - /// - /// `mx_rotate_vector3` is `v·cos θ + (v × axis)·sin θ + axis·(axis·v)(1 − - /// cos θ)`, and `v × axis = −(axis × v)`: Rodrigues' formula at `−θ`. So - /// a `rotate3d` by `amount` degrees turns the tangent by `−amount` in the - /// right-handed sense [`Leaf::rotation`] is stated in. - fn tangent_rotation(&mut self, degrees: Slot) -> Option { - if self.is(degrees, 0.0) { - return None; - } - let k = self.k(-std::f32::consts::PI / 180.0); - Some(self.mul(degrees, k)) - } - - /// `1 − a`. - fn one_minus(&mut self, a: Slot) -> Slot { - let one = self.k(1.0); - self.c.emit(Op::Invert { a, amount: one }) - } - - /// Whether `s` folds to a constant whose every lane is `x`. - fn is(&self, s: Slot, x: f32) -> bool { - self.c.fold(s).is_some_and(|v| { - let lanes = (v.arity as usize).clamp(1, 4); - v.v[..lanes].iter().all(|&c| c == x) - }) - } - - // ---- the tree ---------------------------------------------------------- - - /// A leaf, pruned when its weight folds to zero. - fn leaf( - &mut self, - bsdf: Bsdf, - weight: Slot, - normal: Option, - tangent: Option, - ) -> Option { - if self.is(weight, 0.0) { - return None; - } - if let Bsdf::Sheen { - mode: SheenMode::Zeltner, - .. - } = &bsdf - { - self.out - .reported - .insert("sheen_bsdf mode zeltner (evaluated as conty_kulla)".into()); - } - Some(self.push(Closure::Leaf(Leaf { - bsdf, - weight, - normal, - tangent, - rotation: None, - }))) - } - - /// Turns the tangent of the leaf `id` by `rotation` ([`Leaf::rotation`]). - fn rotate(&mut self, id: Option, rotation: Option) { - if let (Some(id), Some(r)) = (id, rotation) - && let Closure::Leaf(l) = &mut self.out.nodes[id as usize] - { - l.rotation = Some(r); - } - } - - /// The surface's presence, unless it folds to 1. - fn set_opacity(&mut self, opacity: Slot) { - if !self.is(opacity, 1.0) { - self.out.opacity = Some(opacity); - } - } - - fn push(&mut self, c: Closure) -> NodeId { - self.out.nodes.push(c); - (self.out.nodes.len() - 1) as NodeId - } - - fn layer(&mut self, top: Option, base: Option) -> Option { - match (top, base) { - (Some(top), Some(base)) => Some(self.push(Closure::Layer { top, base })), - (t, b) => t.or(b), - } - } - - /// `mix(fg, bg, m)`, pruned at a constant `m` of 0 or 1. A pruned branch - /// keeps its share of the throughput ([`Closures::mix`]). - fn mixc(&mut self, fg: Option, bg: Option, m: Slot) -> Option { - if self.is(m, 0.0) { - return bg; - } - if self.is(m, 1.0) { - return fg; - } - self.out.mix(fg, bg, m) - } - - /// `multiply(x, w)`, pruned at a constant `w` of 0 and elided at 1. - fn multiply(&mut self, x: Option, w: Slot) -> Option { - let x = x?; - if self.is(w, 0.0) { - return None; - } - if self.is(w, 1.0) { - return Some(x); - } - Some(self.push(Closure::Multiply { - input: x, - weight: w, - })) - } - - /// A thin film, `None` when its thickness folds to zero. - fn film(&mut self, thickness_nm: Slot, ior: Slot) -> Option { - (!self.is(thickness_nm, 0.0)).then_some(ThinFilm { - thickness: thickness_nm, - ior, - }) - } - - /// An EDF term, pruned when its weight folds to zero. - fn emit_edf(&mut self, color: Slot, weight: Slot, falloff: Option) { - if self.is(weight, 0.0) || self.is(color, 0.0) { - return; - } - self.out.emission.push(Emission { - color, - weight, - falloff, - }); - } - - // ---- shared helper graphs ---------------------------------------------- - - /// `NG_open_pbr_anisotropy`: `α = r²`, `αx = α·√(2 / (1 + (1 − a)²))`, - /// `αy = (1 − a)·αx`. - fn open_pbr_anisotropy(&mut self, roughness: Slot, anisotropy: Slot) -> Slot { - let inv = self.one_minus(anisotropy); // aniso_invert - let inv_sq = self.mul(inv, inv); // aniso_invert_sq - let one = self.k(1.0); - let denom = self.add(inv_sq, one); // denom - let two = self.k(2.0); - let fraction = self.div(two, denom); // fraction - let sqrt = self.un(UnOp::Sqrt, fraction); // sqrt - let rough_sq = self.mul(roughness, roughness); // rough_sq - let ax = self.mul(rough_sq, sqrt); // alpha_x - let ay = self.mul(inv, ax); // alpha_y - self.c.emit(Op::Combine2 { a: ax, b: ay }) // result - } - - /// `ND_roughness_anisotropy` (`mx_roughness_anisotropy.glsl`): - /// `α = clamp(r², ε, 1)`; with `aspect = √(1 − clamp(a, 0, 0.98))`, - /// `(min(α / aspect, 1), α · aspect)` — which is `(α, α)` at `a ≤ 0`, - /// so the GLSL's branch needs no select here. - fn roughness_anisotropy(&mut self, roughness: Slot, anisotropy: Slot) -> Slot { - let r2 = self.mul(roughness, roughness); - let alpha = self.clamp(r2, 1e-8, 1.0); - let a = self.clamp(anisotropy, 0.0, 0.98); - let one_minus = self.one_minus(a); - let aspect = self.un(UnOp::Sqrt, one_minus); - let x0 = self.div(alpha, aspect); - let one = self.k(1.0); - let x = self.bin(BinOp::Min, x0, one); - let y = self.mul(alpha, aspect); - self.c.emit(Op::Combine2 { a: x, b: y }) - } - - /// `((ior − 1) / (ior + 1))²`. - fn ior_to_f0(&mut self, ior: Slot) -> Slot { - let one = self.k(1.0); - let m = self.sub(ior, one); - let p = self.add(one, ior); - let r = self.div(m, p); - self.mul(r, r) - } -} - -// --------------------------------------------------------------------------- -// open_pbr_surface — NG_open_pbr_surface_surfaceshader (OpenPBR 1.1) -// --------------------------------------------------------------------------- - -fn open_pbr_surface(b: &mut B<'_, '_>) { - b.report( - "transmission_dispersion_scale", - "ignored, as MaterialX's own graph does", - ); - - let normal = b.geom("geometry_normal"); - let tangent = b.geom("geometry_tangent"); - let coat_normal = b.geom("geometry_coat_normal"); - let coat_tangent = b.geom("geometry_coat_tangent"); - - // Roughness: the coat broadens the base specular. - let coat_roughness = b.get("coat_roughness"); - let specular_roughness = b.get("specular_roughness"); - let coat_weight = b.get("coat_weight"); - let four = b.k(4.0); - let cr4 = b.bin(BinOp::Pow, coat_roughness, four); // coat_roughness_to_power_4 - let two = b.k(2.0); - let two_cr4 = b.mul(cr4, two); // two_times_coat_roughness_to_power_4 - let sr4 = b.bin(BinOp::Pow, specular_roughness, four); // specular_roughness_to_power_4 - let sum = b.add(two_cr4, sr4); // add_coat_and_spec_roughnesses_to_power_4 - let one = b.k(1.0); - let min1 = b.bin(BinOp::Min, one, sum); // min_1_add_coat_and_spec_roughnesses_to_power_4 - let quarter = b.k(0.25); - let coat_affected = b.bin(BinOp::Pow, min1, quarter); // coat_affected_specular_roughness - let effective = b.mix(coat_affected, specular_roughness, coat_weight); // effective_specular_roughness - let aniso = b.get("specular_roughness_anisotropy"); - let main_roughness = b.open_pbr_anisotropy(effective, aniso); // main_roughness - - // Subsurface, thin-walled and not, built only when it is live: a - // literal-zero `subsurface_weight` (the default) makes `opaque_base` the - // diffuse alone. - let subsurface_color = b.get("subsurface_color"); - let zero = b.k(0.0); - let diffuse_roughness = b.get("base_diffuse_roughness"); - let one_w = b.k(1.0); - let base_color = b.get("base_color"); - let bcn = b.bin(BinOp::Max, base_color, zero); // base_color_nonnegative - let base_weight = b.get("base_weight"); - let diffuse = b.leaf( - Bsdf::Diffuse { - model: DiffuseModel::Eon, - color: bcn, - roughness: diffuse_roughness, - }, - base_weight, - normal, - None, - ); // diffuse_bsdf - let thin_walled = b.get("geometry_thin_walled"); - let subsurface_weight = b.get("subsurface_weight"); - let opaque_base = if b.is(subsurface_weight, 0.0) { - diffuse - } else { - let ssc = b.bin(BinOp::Max, subsurface_color, zero); // subsurface_color_nonnegative - let sss_refl_bsdf = b.leaf( - Bsdf::Diffuse { - model: DiffuseModel::OrenNayar, - color: ssc, - roughness: diffuse_roughness, - }, - one_w, - normal, - None, - ); // subsurface_thin_walled_reflection_bsdf - let ss_aniso = b.get("subsurface_scatter_anisotropy"); - let one_minus_aniso = b.one_minus(ss_aniso); // one_minus_subsurface_scatter_anisotropy - let brdf_factor = b.mul(subsurface_color, one_minus_aniso); // subsurface_thin_walled_brdf_factor - let sss_refl = b.multiply(sss_refl_bsdf, brdf_factor); // subsurface_thin_walled_reflection - let sss_trans_bsdf = b.leaf(Bsdf::Translucent { color: ssc }, one_w, normal, None); // subsurface_thin_walled_transmission_bsdf - let one = b.k(1.0); - let one_plus_aniso = b.add(one, ss_aniso); // one_plus_subsurface_scatter_anisotropy - let btdf_factor = b.mul(subsurface_color, one_plus_aniso); // subsurface_thin_walled_btdf_factor - let sss_trans = b.multiply(sss_trans_bsdf, btdf_factor); // subsurface_thin_walled_transmission - let half = b.k(0.5); - let sss_thin = b.mixc(sss_refl, sss_trans, half); // subsurface_thin_walled - let radius_scale = b.get("subsurface_radius_scale"); - let radius = b.get("subsurface_radius"); - let radius_scaled = b.mul(radius_scale, radius); // subsurface_radius_scaled - let sss_bsdf = b.leaf( - Bsdf::Subsurface { - color: ssc, - radius: radius_scaled, - anisotropy: ss_aniso, - }, - one_w, - normal, - None, - ); // subsurface_bsdf - let selector = b.convert(thin_walled, 1); // subsurface_selector - let selected = b.mixc(sss_thin, sss_bsdf, selector); // selected_subsurface - b.mixc(selected, diffuse, subsurface_weight) // opaque_base - }; - - // The transmission volume. - let transmission_color = b.get("transmission_color"); - let tcv = b.convert(transmission_color, 3); // transmission_color_vector - let tcl = b.un(UnOp::Ln, tcv); // transmission_color_ln - let minus_one = b.k(-1.0); - let ext_den = b.mul(tcl, minus_one); // extinction_coeff_denom - let depth = b.get("transmission_depth"); - let depth_v = b.convert(depth, 3); // transmission_depth_vector - let extinction = b.div(ext_den, depth_v); // extinction_coeff - let scatter = b.get("transmission_scatter"); - let scatter_v = b.convert(scatter, 3); // transmission_scatter_vector - let scattering = b.div(scatter_v, depth_v); // scattering_coeff - let absorption = b.sub(extinction, scattering); // absorption_coeff - let ax = b.extract(absorption, 0); - let ay = b.extract(absorption, 1); - let az = b.extract(absorption, 2); - let min_xy = b.bin(BinOp::Min, ax, ay); - let amin = b.bin(BinOp::Min, min_xy, az); // absorption_coeff_min - let amin_v = b.convert(amin, 3); - let shifted = b.sub(absorption, amin_v); // absorption_coeff_shifted - let if_shifted = b.gt(zero, amin, shifted, absorption); // if_absorption_coeff_shifted - let zero3 = b.k3(0.0, 0.0, 0.0); - let vol_abs = b.gt(depth, zero, if_shifted, zero3); // if_volume_absorption - let vol_scat = b.gt(depth, zero, scattering, zero3); // if_volume_scattering - let vol_aniso = b.get("transmission_scatter_anisotropy"); - let transmission_weight = b.get("transmission_weight"); - let one = b.k(1.0); - if !b.is(transmission_weight, 0.0) { - b.out.volume = Some(Volume { - absorption: vol_abs, - scattering: vol_scat, - anisotropy: vol_aniso, - }); // dielectric_volume - } - - // The dielectric interface's IOR: relative to the coat, and modulated by - // specular_weight through F0. - let tf_thickness = b.get("thin_film_thickness"); - let thousand = b.k(1000.0); - let tf_nm = b.mul(tf_thickness, thousand); // thin_film_thickness_nm - let specular_ior = b.get("specular_ior"); - let coat_ior = b.get("coat_ior"); - let s2c = b.div(specular_ior, coat_ior); // specular_to_coat_ior_ratio - let c2s = b.div(coat_ior, specular_ior); // coat_to_specular_ior_ratio - let tir_fix = b.gt(s2c, one, s2c, c2s); // specular_to_coat_ior_ratio_tir_fix - let eta_s = b.mix(tir_fix, specular_ior, coat_weight); // eta_s - let em1 = b.sub(eta_s, one); // eta_s_minus_one - let ep1 = b.add(eta_s, one); // eta_s_plus_one - let f0_sqrt = b.div(em1, ep1); // specular_F0_sqrt - let f0 = b.mul(f0_sqrt, f0_sqrt); // specular_F0 - let specular_weight = b.get("specular_weight"); - let scaled_f0 = b.mul(specular_weight, f0); // scaled_specular_F0 - let scaled_f0c = b.clamp(scaled_f0, 0.0, 0.99999); // scaled_specular_F0_clamped - let sqrt_f0 = b.un(UnOp::Sqrt, scaled_f0c); // sqrt_scaled_specular_F0 - let sign = b.un(UnOp::Sign, em1); // sign_eta_s_minus_one - let eps = b.mul(sign, sqrt_f0); // modulated_eta_s_epsilon - let one_minus_eps = b.one_minus(eps); // one_minus_modulated_eta_s_epsilon - let one_plus_eps = b.add(one, eps); // one_plus_modulated_eta_s_epsilon - let modulated_eta = b.div(one_plus_eps, one_minus_eps); // modulated_eta_s - - // Transmission over the opaque base, then the reflection over that. - let white = b.k3(1.0, 1.0, 1.0); - let substrate = if b.is(transmission_weight, 0.0) { - opaque_base - } else { - let t_tint = b.gt(depth, zero, white, transmission_color); // if_transmission_tint - let d_trans = b.leaf( - Bsdf::Dielectric { - tint: t_tint, - ior: modulated_eta, - roughness: main_roughness, - mode: ScatterMode::T, - thin_film: None, - abbe: None, - }, - one_w, - normal, - tangent, - ); // dielectric_transmission (+ dielectric_volume_transmission) - b.mixc(d_trans, opaque_base, transmission_weight) // dielectric_substrate - }; - let specular_color = b.get("specular_color"); - let d_refl = b.leaf( - Bsdf::Dielectric { - tint: specular_color, - ior: modulated_eta, - roughness: main_roughness, - mode: ScatterMode::R, - thin_film: None, - abbe: None, - }, - one_w, - normal, - tangent, - ); // dielectric_reflection - let tf_ior = b.get("thin_film_ior"); - let thin_film_weight = b.get("thin_film_weight"); - let d_refl_mix = if b.is(thin_film_weight, 0.0) { - d_refl - } else { - let film = b.film(tf_nm, tf_ior); - let d_refl_tf = b.leaf( - Bsdf::Dielectric { - tint: specular_color, - ior: modulated_eta, - roughness: main_roughness, - mode: ScatterMode::R, - thin_film: film, - abbe: None, - }, - one_w, - normal, - tangent, - ); // dielectric_reflection_tf - b.mixc(d_refl_tf, d_refl, thin_film_weight) // dielectric_reflection_tf_mix - }; - let dielectric_base = b.layer(d_refl_mix, substrate); // dielectric_base - - // The metal. - let metal_refl = b.mul(base_color, base_weight); // metal_reflectivity - let metal_edge = b.mul(specular_color, specular_weight); // metal_edgecolor - let five = b.k(5.0); - let metalness = b.get("base_metalness"); - let metal_mix = if b.is(metalness, 0.0) { - None - } else { - let metal = b.leaf( - Bsdf::Schlick { - color0: metal_refl, - color82: metal_edge, - color90: white, - exponent: five, - roughness: main_roughness, - mode: ScatterMode::R, - thin_film: None, - }, - specular_weight, - normal, - tangent, - ); // metal_bsdf - if b.is(thin_film_weight, 0.0) { - metal - } else { - let film = b.film(tf_nm, tf_ior); - let metal_tf = b.leaf( - Bsdf::Schlick { - color0: metal_refl, - color82: metal_edge, - color90: white, - exponent: five, - roughness: main_roughness, - mode: ScatterMode::R, - thin_film: film, - }, - specular_weight, - normal, - tangent, - ); // metal_bsdf_tf - b.mixc(metal_tf, metal, thin_film_weight) // metal_bsdf_tf_mix - } - }; - let base_substrate = b.mixc(metal_mix, dielectric_base, metalness); // base_substrate - - // Coat darkening and tint. - let coat_f0 = b.ior_to_f0(coat_ior); // coat_ior_to_F0 - let one_minus_coat_f0 = b.one_minus(coat_f0); // one_minus_coat_F0 - let coat_ior_sq = b.mul(coat_ior, coat_ior); // coat_ior_sqr - let omf0_eta2 = b.div(one_minus_coat_f0, coat_ior_sq); // one_minus_coat_F0_over_eta2 - let k_coat = b.one_minus(omf0_eta2); // Kcoat - let e_metal = b.mul(base_color, specular_weight); // Emetal - let e_diel = b.mix(subsurface_color, base_color, subsurface_weight); // Edielectric - let e_base = b.mix(e_metal, e_diel, metalness); // Ebase - let ebk = b.mul(e_base, k_coat); // Ebase_Kcoat - let one_minus_k = b.one_minus(k_coat); // one_minus_Kcoat - let one_minus_ebk = b.sub(white, ebk); // one_minus_Ebase_Kcoat - let omk3 = b.convert(one_minus_k, 3); // one_minus_Kcoat_color - let darkening = b.div(omk3, one_minus_ebk); // base_darkening - let coat_darkening = b.get("coat_darkening"); - let cwd = b.mul(coat_weight, coat_darkening); // coat_weight_times_coat_darkening - let mod_dark = b.mix(darkening, white, cwd); // modulated_base_darkening - let darkened = b.multiply(base_substrate, mod_dark); // darkened_base_substrate - let coat_color = b.get("coat_color"); - let coat_att = b.mix(coat_color, white, coat_weight); // coat_attenuation - let attenuated = b.multiply(darkened, coat_att); // coat_substrate_attenuated - let coat_aniso = b.get("coat_roughness_anisotropy"); - let coat_rough_v = b.open_pbr_anisotropy(coat_roughness, coat_aniso); // coat_roughness_vector - let coat = b.leaf( - Bsdf::Dielectric { - tint: white, - ior: coat_ior, - roughness: coat_rough_v, - mode: ScatterMode::R, - thin_film: None, - abbe: None, - }, - coat_weight, - coat_normal, - coat_tangent, - ); // coat_bsdf - let coat_layer = b.layer(coat, attenuated); // coat_layer - let fuzz_color = b.get("fuzz_color"); - let fuzz_roughness = b.get("fuzz_roughness"); - let fuzz_weight = b.get("fuzz_weight"); - let fuzz = b.leaf( - Bsdf::Sheen { - color: fuzz_color, - roughness: fuzz_roughness, - mode: SheenMode::Zeltner, - }, - fuzz_weight, - normal, - None, - ); // fuzz_bsdf - b.out.root = b.layer(fuzz, coat_layer); // fuzz_layer - - // Emission: uncoated, and through the coat's Fresnel. - let emission_color = b.get("emission_color"); - let luminance = b.get("emission_luminance"); - let ew = b.mul(emission_color, luminance); // emission_weight - let uncoated_w = b.one_minus(coat_weight); - b.emit_edf(ew, uncoated_w, None); // uncoated_emission_edf (bg of emission_edf) - let coated_w = b.mul(coat_color, coat_weight); // coat_tinted_emission_edf · mix - let c0 = b.convert(one_minus_coat_f0, 3); // one_minus_coat_F0_color - let falloff = EdfFalloff { - color0: c0, - color90: zero3, - exponent: five, - }; - b.emit_edf(ew, coated_w, Some(falloff)); // coated_emission_edf (fg) - - b.out.thin_walled = Some(thin_walled); - let opacity = b.get("geometry_opacity"); - b.set_opacity(opacity); // shader_constructor.opacity -} - -// --------------------------------------------------------------------------- -// standard_surface — NG_standard_surface_surfaceshader_100 -// --------------------------------------------------------------------------- - -fn standard_surface(b: &mut B<'_, '_>) { - for (name, why) in [ - ( - "transmission_depth", - "ignored, as MaterialX's own graph does", - ), - ( - "transmission_scatter", - "ignored, as MaterialX's own graph does", - ), - ( - "transmission_dispersion", - "ignored, as MaterialX's own graph does", - ), - ] { - b.report(name, why); - } - - let normal = b.geom("normal"); - let coat_normal = b.geom("coat_normal"); - let tangent = b.geom("tangent"); - // `main_tangent` / `coat_tangent`: the tangent turned by a fraction of a - // full turn about `normal` / `coat_normal`, only where the lobe is - // anisotropic — the leaves' own normals, so a leaf rotation. - let main_rotation = standard_rotation(b, "specular_rotation", "specular_anisotropy"); - let coat_rotation = standard_rotation(b, "coat_rotation", "coat_anisotropy"); - - let coat_affect_roughness = b.get("coat_affect_roughness"); - let coat = b.get("coat"); - let coat_roughness = b.get("coat_roughness"); - let m1 = b.mul(coat_affect_roughness, coat); // coat_affect_roughness_multiply1 - let m2 = b.mul(m1, coat_roughness); // coat_affect_roughness_multiply2 - let specular_roughness = b.get("specular_roughness"); - let one = b.k(1.0); - let coat_affected = b.mix(one, specular_roughness, m2); // coat_affected_roughness - let specular_anisotropy = b.get("specular_anisotropy"); - let main_roughness = b.roughness_anisotropy(coat_affected, specular_anisotropy); // main_roughness - let extra = b.get("transmission_extra_roughness"); - let tr_add = b.add(specular_roughness, extra); // transmission_roughness_add - let tr_clamped = b.clamp(tr_add, 0.0, 1.0); // transmission_roughness_clamped - let coat_affected_tr = b.mix(one, tr_clamped, m2); // coat_affected_transmission_roughness - let transmission_roughness = b.roughness_anisotropy(coat_affected_tr, specular_anisotropy); // transmission_roughness - - let coat_clamped = b.clamp(coat, 0.0, 1.0); // coat_clamped - let coat_affect_color = b.get("coat_affect_color"); - let cg_m = b.mul(coat_clamped, coat_affect_color); // coat_gamma_multiply - let coat_gamma = b.add(cg_m, one); // coat_gamma - let zero = b.k(0.0); - let base_color = b.get("base_color"); - let bcn = b.bin(BinOp::Max, base_color, zero); // base_color_nonnegative - let diffuse_color = b.bin(BinOp::Pow, bcn, coat_gamma); // coat_affected_diffuse_color - let subsurface_color = b.get("subsurface_color"); - let scn = b.bin(BinOp::Max, subsurface_color, zero); // subsurface_color_nonnegative - let sss_color = b.bin(BinOp::Pow, scn, coat_gamma); // coat_affected_subsurface_color - - let base = b.get("base"); - let diffuse_roughness = b.get("diffuse_roughness"); - let diffuse = b.leaf( - Bsdf::Diffuse { - model: DiffuseModel::OrenNayar, - color: diffuse_color, - roughness: diffuse_roughness, - }, - base, - normal, - None, - ); // diffuse_bsdf - let subsurface = b.get("subsurface"); - let sss_mix = if b.is(subsurface, 0.0) { - diffuse - } else { - let one_w = b.k(1.0); - let translucent = b.leaf(Bsdf::Translucent { color: sss_color }, one_w, normal, None); // translucent_bsdf - let radius = b.get("subsurface_radius"); - let scale = b.get("subsurface_scale"); - let radius_scaled = b.mul(radius, scale); // subsurface_radius_scaled - let ss_aniso = b.get("subsurface_anisotropy"); - let sss = b.leaf( - Bsdf::Subsurface { - color: sss_color, - radius: radius_scaled, - anisotropy: ss_aniso, - }, - one_w, - normal, - None, - ); // subsurface_bsdf - let thin_walled = b.get("thin_walled"); - let selector = b.convert(thin_walled, 1); // subsurface_selector - let selected = b.mixc(translucent, sss, selector); // selected_subsurface_bsdf - b.mixc(selected, diffuse, subsurface) // subsurface_mix - }; - let sheen_w = b.get("sheen"); - let sheen_color = b.get("sheen_color"); - let sheen_roughness = b.get("sheen_roughness"); - let sheen = b.leaf( - Bsdf::Sheen { - color: sheen_color, - roughness: sheen_roughness, - mode: SheenMode::ContyKulla, - }, - sheen_w, - normal, - None, - ); // sheen_bsdf - let sheen_layer = b.layer(sheen, sss_mix); // sheen_layer - let transmission = b.get("transmission"); - let transmission_color = b.get("transmission_color"); - let specular_ior = b.get("specular_IOR"); - let one_w = b.k(1.0); - let trans_mix = if b.is(transmission, 0.0) { - sheen_layer - } else { - let trans = b.leaf( - Bsdf::Dielectric { - tint: transmission_color, - ior: specular_ior, - roughness: transmission_roughness, - mode: ScatterMode::T, - thin_film: None, - abbe: None, - }, - one_w, - normal, - tangent, - ); // transmission_bsdf - b.rotate(trans, main_rotation); - b.mixc(trans, sheen_layer, transmission) // transmission_mix - }; - let specular = b.get("specular"); - let specular_color = b.get("specular_color"); - let tf_thickness = b.get("thin_film_thickness"); - let tf_ior = b.get("thin_film_IOR"); - let film = b.film(tf_thickness, tf_ior); - let spec = b.leaf( - Bsdf::Dielectric { - tint: specular_color, - ior: specular_ior, - roughness: main_roughness, - mode: ScatterMode::R, - thin_film: film, - abbe: None, - }, - specular, - normal, - tangent, - ); // specular_bsdf - b.rotate(spec, main_rotation); - let spec_layer = b.layer(spec, trans_mix); // specular_layer - let metalness = b.get("metalness"); - let metal_mix = if b.is(metalness, 0.0) { - spec_layer - } else { - let metal_refl = b.mul(base_color, base); // metal_reflectivity - let metal_edge = b.mul(specular_color, specular); // metal_edgecolor - let n = b.c.emit(Op::ArtisticIor { - reflectivity: metal_refl, - edge: metal_edge, - extinction: false, - }); // artistic_ior.ior - let k = b.c.emit(Op::ArtisticIor { - reflectivity: metal_refl, - edge: metal_edge, - extinction: true, - }); // artistic_ior.extinction - let metal = b.leaf( - Bsdf::Conductor { - ior: n, - extinction: k, - roughness: main_roughness, - thin_film: film, - }, - one_w, - normal, - tangent, - ); // metal_bsdf - b.rotate(metal, main_rotation); - b.mixc(metal, spec_layer, metalness) // metalness_mix - }; - let coat_color = b.get("coat_color"); - let white = b.k3(1.0, 1.0, 1.0); - let coat_att = b.mix(coat_color, white, coat); // coat_attenuation - let attenuated = b.multiply(metal_mix, coat_att); // thin_film_layer_attenuated - let coat_anisotropy = b.get("coat_anisotropy"); - let coat_rough_v = b.roughness_anisotropy(coat_roughness, coat_anisotropy); // coat_roughness_vector - let coat_ior = b.get("coat_IOR"); - let coat_bsdf = b.leaf( - Bsdf::Dielectric { - tint: white, - ior: coat_ior, - roughness: coat_rough_v, - mode: ScatterMode::R, - thin_film: None, - abbe: None, - }, - coat, - coat_normal, - tangent, - ); // coat_bsdf - b.rotate(coat_bsdf, coat_rotation); - b.out.root = b.layer(coat_bsdf, attenuated); // coat_layer - - // Emission, uncoated and through the coat's Fresnel. - let coat_f0 = b.ior_to_f0(coat_ior); // coat_ior_to_F0 - let one_minus_f0 = b.one_minus(coat_f0); // one_minus_coat_ior_to_F0 - let emission_color = b.get("emission_color"); - let emission = b.get("emission"); - let ew = b.mul(emission_color, emission); // emission_weight - let uncoated_w = b.one_minus(coat); - b.emit_edf(ew, uncoated_w, None); // emission_edf (bg of blended_coat_emission_edf) - let coated_w = b.mul(coat_color, coat); // coat_tinted_emission_edf · mix - let c0 = b.convert(one_minus_f0, 3); // emission_color0 - let zero3 = b.k3(0.0, 0.0, 0.0); - let five = b.k(5.0); - b.emit_edf( - ew, - coated_w, - Some(EdfFalloff { - color0: c0, - color90: zero3, - exponent: five, - }), - ); // coat_emission_edf (fg) - - let thin_walled = b.get("thin_walled"); - b.out.thin_walled = Some(thin_walled); - let opacity = b.get("opacity"); - let luminance = b.c.luminance(opacity); // opacity_luminance(_float) - b.set_opacity(luminance); // shader_constructor.opacity -} - -/// `standard_surface`'s `main_tangent` / `coat_tangent`: `rotate3d` by -/// `rotation · 360` degrees, selected by `ifgreater(anisotropy, 0)`. -fn standard_rotation( - b: &mut B<'_, '_>, - rotation: &'static str, - anisotropy: &'static str, -) -> Option { - let r = b.get(rotation); - let full_turn = b.k(360.0); - let degrees = b.mul(r, full_turn); // tangent_rotate_degree - let angle = b.tangent_rotation(degrees)?; // tangent_rotate - let a = b.get(anisotropy); - let zero = b.k(0.0); - let selected = b.gt(a, zero, angle, zero); // main_tangent / coat_tangent - (!b.is(selected, 0.0)).then_some(selected) -} - -// --------------------------------------------------------------------------- -// gltf_pbr — IMPL_gltf_pbr_surfaceshader (glTF PBR 2.0.1) -// --------------------------------------------------------------------------- - -fn gltf_pbr(b: &mut B<'_, '_>) { - b.report("occlusion", "a path tracer computes its own"); - b.report("dispersion", "ignored, as MaterialX's own graph does"); - b.report("thickness", "ignored, as MaterialX's own graph does"); - - let normal = b.geom("normal"); - let tangent = b.geom("tangent"); - let clearcoat_normal = b.geom("clearcoat_normal"); - // `selected_tangent`: the tangent turned by `anisotropy_rotation` radians - // about `normal`, for every base leaf; the clearcoat keeps `tangent`. - // The graph's `ifgreater(|rotation|, 0)` needs no select here, since a - // zero angle turns nothing. - let aniso_rotation = b.get("anisotropy_rotation"); - let to_degrees = b.k(-57.29578); - let degrees = b.mul(aniso_rotation, to_degrees); // rad_2_deg - let rotation = b.tangent_rotation(degrees); // rotate_tangent - - // The volume. - let transmission = b.get("transmission"); - if !b.is(transmission, 0.0) { - let ac = b.get("attenuation_color"); - let acv = b.convert(ac, 3); // attenuation_color_vec - let ln = b.un(UnOp::Ln, acv); // ln_attenuation_color_vec - let dist = b.get("attenuation_distance"); - let zero = b.k(0.0); - let one = b.k(1.0); - let safe = b.gt(dist, zero, dist, one); // safe_attenuation_distance - let over = b.div(ln, safe); // ln_attenuation_color_vec_over_distance - let minus_one = b.k(-1.0); - let coeff = b.mul(over, minus_one); // attenuation_coeff - let zero3 = b.k3(0.0, 0.0, 0.0); - let g = b.k(0.0); - b.out.volume = Some(Volume { - absorption: coeff, - scattering: zero3, - anisotropy: g, - }); // isotropic_volume - } - - // The dielectric's Fresnel as a generalized Schlick. - let ior = b.get("ior"); - let f0_ior = b.ior_to_f0(ior); // dielectric_f0_from_ior - let specular_color = b.get("specular_color"); - let f0_sc = b.mul(specular_color, f0_ior); // dielectric_f0_from_ior_specular_color - let one = b.k(1.0); - let f0_cl = b.bin(BinOp::Min, f0_sc, one); // clamped_dielectric_f0_from_ior_specular_color - let specular = b.get("specular"); - let f0 = b.mul(f0_cl, specular); // dielectric_f0 - let white = b.k3(1.0, 1.0, 1.0); - let f90 = b.mul(white, specular); // dielectric_f90 - - // Roughness. - let roughness = b.get("roughness"); - let alpha = b.mul(roughness, roughness); // alpha_roughness - let strength = b.get("anisotropy_strength"); - let s2 = b.mul(strength, strength); // strength_2 - let at = b.mix(one, alpha, s2); // at - let at_c = b.clamp(at, 0.00001, 1.0); // clamped_at - let ab_c = b.clamp(alpha, 0.00001, 1.0); // clamped_ab - let ruv = b.c.emit(Op::Combine2 { a: at_c, b: ab_c }); // roughness_uv - - let base_color = b.get("base_color"); - let zero = b.k(0.0); - let one_w = b.k(1.0); - let diffuse = b.leaf( - Bsdf::Diffuse { - model: DiffuseModel::OrenNayar, - color: base_color, - roughness: zero, - }, - one_w, - normal, - None, - ); // diffuse_bsdf - let trans_mix = if b.is(transmission, 0.0) { - diffuse - } else { - let trans = b.leaf( - Bsdf::Dielectric { - tint: base_color, - ior, - roughness: ruv, - mode: ScatterMode::T, - thin_film: None, - abbe: None, - }, - one_w, - normal, - tangent, - ); // transmission_bsdf (+ volume_transmission_bsdf) - b.rotate(trans, rotation); - b.mixc(trans, diffuse, transmission) // transmission_mix - }; - let five = b.k(5.0); - let refl = b.leaf( - Bsdf::Schlick { - color0: f0, - color82: white, - color90: f90, - exponent: five, - roughness: ruv, - mode: ScatterMode::R, - thin_film: None, - }, - one_w, - normal, - tangent, - ); // reflection_bsdf - b.rotate(refl, rotation); - let iridescence = b.get("iridescence"); - let irid_thickness = b.get("iridescence_thickness"); - let irid_ior = b.get("iridescence_ior"); - let mix_irid = if b.is(iridescence, 0.0) { - refl - } else { - let film = b.film(irid_thickness, irid_ior); - let tf_refl = b.leaf( - Bsdf::Schlick { - color0: f0, - color82: white, - color90: f90, - exponent: five, - roughness: ruv, - mode: ScatterMode::R, - thin_film: film, - }, - one_w, - normal, - tangent, - ); // tf_reflection_bsdf - b.rotate(tf_refl, rotation); - b.mixc(tf_refl, refl, iridescence) // mix_iridescent_dielectric_reflection - }; - let irid_diel = b.layer(mix_irid, trans_mix); // iridescent_dielectric_bsdf - let metallic = b.get("metallic"); - let metal_mix = if b.is(metallic, 0.0) { - None - } else { - let metal = b.leaf( - Bsdf::Schlick { - color0: base_color, - color82: white, - color90: white, - exponent: five, - roughness: ruv, - mode: ScatterMode::R, - thin_film: None, - }, - one_w, - normal, - tangent, - ); // metal_bsdf - b.rotate(metal, rotation); - if b.is(iridescence, 0.0) { - metal - } else { - let film = b.film(irid_thickness, irid_ior); - let tf_metal = b.leaf( - Bsdf::Schlick { - color0: base_color, - color82: white, - color90: white, - exponent: five, - roughness: ruv, - mode: ScatterMode::R, - thin_film: film, - }, - one_w, - normal, - tangent, - ); // tf_metal_bsdf - b.rotate(tf_metal, rotation); - b.mixc(tf_metal, metal, iridescence) // mix_iridescent_metal_bsdf - } - }; - let base_mix = b.mixc(metal_mix, irid_diel, metallic); // base_mix - - let sheen_color = b.get("sheen_color"); - let r = b.extract(sheen_color, 0); - let g = b.extract(sheen_color, 1); - let bl = b.extract(sheen_color, 2); - let max_rg = b.bin(BinOp::Max, r, g); // sheen_color_max_rg - let intensity = b.bin(BinOp::Max, max_rg, bl); // sheen_intensity - let sheen_roughness = b.get("sheen_roughness"); - let sheen_rough_sq = b.mul(sheen_roughness, sheen_roughness); // sheen_roughness_sq - let sheen_norm = b.div(sheen_color, intensity); // sheen_color_normalized - let sheen = b.leaf( - Bsdf::Sheen { - color: sheen_norm, - roughness: sheen_rough_sq, - mode: SheenMode::ContyKulla, - }, - intensity, - normal, - None, - ); // sheen_bsdf - let sheen_layer = b.layer(sheen, base_mix); // sheen_layer - - let cc_roughness = b.get("clearcoat_roughness"); - let cc_rough_v = b.roughness_anisotropy(cc_roughness, zero); // clearcoat_roughness_uv - let clearcoat = b.get("clearcoat"); - let cc_ior = b.k(1.5); - let cc = b.leaf( - Bsdf::Dielectric { - tint: white, - ior: cc_ior, - roughness: cc_rough_v, - mode: ScatterMode::R, - thin_film: None, - abbe: None, - }, - clearcoat, - clearcoat_normal, - tangent, - ); // clearcoat_bsdf - b.out.root = b.layer(cc, sheen_layer); // clearcoat_layer - - let emissive = b.get("emissive"); - let strength_e = b.get("emissive_strength"); - let ec = b.mul(emissive, strength_e); // emission_color - let one_e = b.k(1.0); - b.emit_edf(ec, one_e, None); // emission - - // Alpha. `alpha_mode` is a uniform, so it folds, and the graph's two - // `ifequal`s select one branch outright: OPAQUE never reads `alpha`, - // which exporters connect to a texture's alpha whatever the mode, and - // folding the select through that texture is not possible. Nor is it - // compiled there: compiling it would load that texture, or report its - // unsupported nodes, for an input the surface cannot show. - let mode = b.get("alpha_mode"); - let folded = b.c.fold(mode).map(|v| v.x()); - if folded == Some(0.0) { - return; // OPAQUE - } - let alpha = b.get("alpha"); - let opacity = match folded { - Some(1.0) => alpha_mask(b, alpha), // MASK - Some(_) => alpha, // BLEND - None => { - let masked = alpha_mask(b, alpha); - let mask_mode = b.k(1.0); - let mask = b.eq(mode, mask_mode, masked, alpha); // opacity_mask - let (opaque_mode, one) = (b.k(0.0), b.k(1.0)); - b.eq(mode, opaque_mode, one, mask) // opacity - } - }; - b.set_opacity(opacity); // shader_constructor.opacity -} - -/// glTF's `opacity_mask_cutoff`: 1 where `alpha ≥ alpha_cutoff`, else 0. -fn alpha_mask(b: &mut B<'_, '_>, alpha: Slot) -> Slot { - let cutoff = b.get("alpha_cutoff"); - let (one, zero) = (b.k(1.0), b.k(0.0)); - b.ge(alpha, cutoff, one, zero) -} diff --git a/crates/crust-mtlx/src/surface/gltf_pbr.rs b/crates/crust-mtlx/src/surface/gltf_pbr.rs new file mode 100644 index 00000000..444f62b5 --- /dev/null +++ b/crates/crust-mtlx/src/surface/gltf_pbr.rs @@ -0,0 +1,259 @@ +//! gltf_pbr — IMPL_gltf_pbr_surfaceshader (glTF PBR 2.0.1) + +use super::*; + +pub(super) fn gltf_pbr(b: &mut B<'_, '_>) { + b.report("occlusion", "a path tracer computes its own"); + b.report("dispersion", "ignored, as MaterialX's own graph does"); + b.report("thickness", "ignored, as MaterialX's own graph does"); + + let normal = b.geom("normal"); + let tangent = b.geom("tangent"); + let clearcoat_normal = b.geom("clearcoat_normal"); + // `selected_tangent`: the tangent turned by `anisotropy_rotation` radians + // about `normal`, for every base leaf; the clearcoat keeps `tangent`. + // The graph's `ifgreater(|rotation|, 0)` needs no select here, since a + // zero angle turns nothing. + let aniso_rotation = b.get("anisotropy_rotation"); + let to_degrees = b.k(-57.29578); + let degrees = b.mul(aniso_rotation, to_degrees); // rad_2_deg + let rotation = b.tangent_rotation(degrees); // rotate_tangent + + // The volume. + let transmission = b.get("transmission"); + if !b.is(transmission, 0.0) { + let ac = b.get("attenuation_color"); + let acv = b.convert(ac, 3); // attenuation_color_vec + let ln = b.un(UnOp::Ln, acv); // ln_attenuation_color_vec + let dist = b.get("attenuation_distance"); + let zero = b.k(0.0); + let one = b.k(1.0); + let safe = b.gt(dist, zero, dist, one); // safe_attenuation_distance + let over = b.div(ln, safe); // ln_attenuation_color_vec_over_distance + let minus_one = b.k(-1.0); + let coeff = b.mul(over, minus_one); // attenuation_coeff + let zero3 = b.k3(0.0, 0.0, 0.0); + let g = b.k(0.0); + b.out.volume = Some(Volume { + absorption: coeff, + scattering: zero3, + anisotropy: g, + }); // isotropic_volume + } + + // The dielectric's Fresnel as a generalized Schlick. + let ior = b.get("ior"); + let f0_ior = b.ior_to_f0(ior); // dielectric_f0_from_ior + let specular_color = b.get("specular_color"); + let f0_sc = b.mul(specular_color, f0_ior); // dielectric_f0_from_ior_specular_color + let one = b.k(1.0); + let f0_cl = b.bin(BinOp::Min, f0_sc, one); // clamped_dielectric_f0_from_ior_specular_color + let specular = b.get("specular"); + let f0 = b.mul(f0_cl, specular); // dielectric_f0 + let white = b.k3(1.0, 1.0, 1.0); + let f90 = b.mul(white, specular); // dielectric_f90 + + // Roughness. + let roughness = b.get("roughness"); + let alpha = b.mul(roughness, roughness); // alpha_roughness + let strength = b.get("anisotropy_strength"); + let s2 = b.mul(strength, strength); // strength_2 + let at = b.mix(one, alpha, s2); // at + let at_c = b.clamp(at, 0.00001, 1.0); // clamped_at + let ab_c = b.clamp(alpha, 0.00001, 1.0); // clamped_ab + let ruv = b.c.emit(Op::Combine2 { a: at_c, b: ab_c }); // roughness_uv + + let base_color = b.get("base_color"); + let zero = b.k(0.0); + let one_w = b.k(1.0); + let diffuse = b.leaf( + Bsdf::Diffuse { + model: DiffuseModel::OrenNayar, + color: base_color, + roughness: zero, + }, + one_w, + normal, + None, + ); // diffuse_bsdf + let trans_mix = if b.is(transmission, 0.0) { + diffuse + } else { + let trans = b.leaf( + Bsdf::Dielectric { + tint: base_color, + ior, + roughness: ruv, + mode: ScatterMode::T, + thin_film: None, + abbe: None, + }, + one_w, + normal, + tangent, + ); // transmission_bsdf (+ volume_transmission_bsdf) + b.rotate(trans, rotation); + b.mixc(trans, diffuse, transmission) // transmission_mix + }; + let five = b.k(5.0); + let refl = b.leaf( + Bsdf::Schlick { + color0: f0, + color82: white, + color90: f90, + exponent: five, + roughness: ruv, + mode: ScatterMode::R, + thin_film: None, + }, + one_w, + normal, + tangent, + ); // reflection_bsdf + b.rotate(refl, rotation); + let iridescence = b.get("iridescence"); + let irid_thickness = b.get("iridescence_thickness"); + let irid_ior = b.get("iridescence_ior"); + let mix_irid = if b.is(iridescence, 0.0) { + refl + } else { + let film = b.film(irid_thickness, irid_ior); + let tf_refl = b.leaf( + Bsdf::Schlick { + color0: f0, + color82: white, + color90: f90, + exponent: five, + roughness: ruv, + mode: ScatterMode::R, + thin_film: film, + }, + one_w, + normal, + tangent, + ); // tf_reflection_bsdf + b.rotate(tf_refl, rotation); + b.mixc(tf_refl, refl, iridescence) // mix_iridescent_dielectric_reflection + }; + let irid_diel = b.layer(mix_irid, trans_mix); // iridescent_dielectric_bsdf + let metallic = b.get("metallic"); + let metal_mix = if b.is(metallic, 0.0) { + None + } else { + let metal = b.leaf( + Bsdf::Schlick { + color0: base_color, + color82: white, + color90: white, + exponent: five, + roughness: ruv, + mode: ScatterMode::R, + thin_film: None, + }, + one_w, + normal, + tangent, + ); // metal_bsdf + b.rotate(metal, rotation); + if b.is(iridescence, 0.0) { + metal + } else { + let film = b.film(irid_thickness, irid_ior); + let tf_metal = b.leaf( + Bsdf::Schlick { + color0: base_color, + color82: white, + color90: white, + exponent: five, + roughness: ruv, + mode: ScatterMode::R, + thin_film: film, + }, + one_w, + normal, + tangent, + ); // tf_metal_bsdf + b.rotate(tf_metal, rotation); + b.mixc(tf_metal, metal, iridescence) // mix_iridescent_metal_bsdf + } + }; + let base_mix = b.mixc(metal_mix, irid_diel, metallic); // base_mix + + let sheen_color = b.get("sheen_color"); + let r = b.extract(sheen_color, 0); + let g = b.extract(sheen_color, 1); + let bl = b.extract(sheen_color, 2); + let max_rg = b.bin(BinOp::Max, r, g); // sheen_color_max_rg + let intensity = b.bin(BinOp::Max, max_rg, bl); // sheen_intensity + let sheen_roughness = b.get("sheen_roughness"); + let sheen_rough_sq = b.mul(sheen_roughness, sheen_roughness); // sheen_roughness_sq + let sheen_norm = b.div(sheen_color, intensity); // sheen_color_normalized + let sheen = b.leaf( + Bsdf::Sheen { + color: sheen_norm, + roughness: sheen_rough_sq, + mode: SheenMode::ContyKulla, + }, + intensity, + normal, + None, + ); // sheen_bsdf + let sheen_layer = b.layer(sheen, base_mix); // sheen_layer + + let cc_roughness = b.get("clearcoat_roughness"); + let cc_rough_v = b.roughness_anisotropy(cc_roughness, zero); // clearcoat_roughness_uv + let clearcoat = b.get("clearcoat"); + let cc_ior = b.k(1.5); + let cc = b.leaf( + Bsdf::Dielectric { + tint: white, + ior: cc_ior, + roughness: cc_rough_v, + mode: ScatterMode::R, + thin_film: None, + abbe: None, + }, + clearcoat, + clearcoat_normal, + tangent, + ); // clearcoat_bsdf + b.out.root = b.layer(cc, sheen_layer); // clearcoat_layer + + let emissive = b.get("emissive"); + let strength_e = b.get("emissive_strength"); + let ec = b.mul(emissive, strength_e); // emission_color + let one_e = b.k(1.0); + b.emit_edf(ec, one_e, None); // emission + + // Alpha. `alpha_mode` is a uniform, so it folds, and the graph's two + // `ifequal`s select one branch outright: OPAQUE never reads `alpha`, + // which exporters connect to a texture's alpha whatever the mode, and + // folding the select through that texture is not possible. Nor is it + // compiled there: compiling it would load that texture, or report its + // unsupported nodes, for an input the surface cannot show. + let mode = b.get("alpha_mode"); + let folded = b.c.fold(mode).map(|v| v.x()); + if folded == Some(0.0) { + return; // OPAQUE + } + let alpha = b.get("alpha"); + let opacity = match folded { + Some(1.0) => alpha_mask(b, alpha), // MASK + Some(_) => alpha, // BLEND + None => { + let masked = alpha_mask(b, alpha); + let mask_mode = b.k(1.0); + let mask = b.eq(mode, mask_mode, masked, alpha); // opacity_mask + let (opaque_mode, one) = (b.k(0.0), b.k(1.0)); + b.eq(mode, opaque_mode, one, mask) // opacity + } + }; + b.set_opacity(opacity); // shader_constructor.opacity +} + +/// glTF's `opacity_mask_cutoff`: 1 where `alpha ≥ alpha_cutoff`, else 0. +fn alpha_mask(b: &mut B<'_, '_>, alpha: Slot) -> Slot { + let cutoff = b.get("alpha_cutoff"); + let (one, zero) = (b.k(1.0), b.k(0.0)); + b.ge(alpha, cutoff, one, zero) +} diff --git a/crates/crust-mtlx/src/surface/mod.rs b/crates/crust-mtlx/src/surface/mod.rs new file mode 100644 index 00000000..f022db5a --- /dev/null +++ b/crates/crust-mtlx/src/surface/mod.rs @@ -0,0 +1,525 @@ +//! MaterialX surface-shader nodes expanded into the closure tree of their +//! implementation nodegraphs. +//! +//! `open_pbr_surface`, `standard_surface` and `gltf_pbr` are, in MaterialX, +//! nodedefs whose implementations are nodegraphs over standalone BSDF nodes +//! (`libraries/bxdf/*.mtlx`, MaterialX 1.39). A document authoring one never +//! carries the implementation, so this module reproduces each graph **node for +//! node** — the same leaves, the same `layer` / `mix` / `multiply` in the same +//! order, and the same derived parameters, which are emitted as program ops so +//! they fold, optimise and JIT like any pattern graph. Each block below is +//! commented with the nodegraph node names it reproduces, so it can be checked +//! against the `.mtlx` line by line. NVIDIA Typhoon builds its surfaces the +//! same way (`MaterialXCpp/materials/*.cpp`); where Typhoon departs from the +//! graph (it omits OpenPBR's thin-walled subsurface branch), the graph wins. +//! +//! Every input takes its connection, else its authored value, else its +//! nodedef default from the tables here — generated from the nodedefs, which +//! are vendored under `tests/nodedefs/` and checked against these tables. +//! +//! Two parts of each graph sit outside the closure tree and are carried beside +//! it: the `surface` node's `opacity` ([`Closures::opacity`]), and the +//! `rotate3d` a graph applies to its tangent, which becomes the angle a leaf's +//! frame is turned by ([`Leaf::rotation`]). +//! +//! What the tree cannot represent is reported rather than dropped silently: +//! glTF occlusion, and inputs the MaterialX graphs themselves ignore. + +use crate::bsdf::{ + Bsdf, Closure, Closures, DiffuseModel, EdfFalloff, Emission, Leaf, NodeId, ScatterMode, + SheenMode, Slot, ThinFilm, Volume, +}; +use crate::eval::{BinOp, Compiler, Op, UnOp}; +use crate::parse::Node; +use crate::value::Val; +use std::collections::HashMap; + +mod gltf_pbr; +mod open_pbr; +mod standard_surface; + +use gltf_pbr::gltf_pbr; +use open_pbr::open_pbr_surface; +use standard_surface::standard_surface; + +/// One nodedef input: its name, MaterialX type and default. `None` for an +/// input whose default is a geometric property (`Nworld`, `Tworld`) or that +/// declares no value. +#[derive(Clone, Copy, Debug)] +pub struct InputDef { + pub name: &'static str, + pub ty: &'static str, + pub default: Option, +} + +// Generated from `tests/nodedefs/*.mtlx` (MaterialX 1.39, Apache-2.0); the +// `nodedef_tables_match_materialx` test keeps them honest. `standard_surface` +// is version 1.0.1, which inherits 1.0.0 and overrides `base` and +// `base_color`. +#[rustfmt::skip] +mod tables { + use super::InputDef; + use crate::value::Val; + /// `ND_open_pbr_surface_surfaceshader`'s inputs, in nodedef order. + pub const OPEN_PBR_SURFACE: &[InputDef] = &[ + InputDef { name: "base_weight", ty: "float", default: Some(Val::float(1.0)) }, + InputDef { name: "base_color", ty: "color3", default: Some(Val::vec3(0.8, 0.8, 0.8)) }, + InputDef { name: "base_diffuse_roughness", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "base_metalness", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "specular_weight", ty: "float", default: Some(Val::float(1.0)) }, + InputDef { name: "specular_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, + InputDef { name: "specular_roughness", ty: "float", default: Some(Val::float(0.3)) }, + InputDef { name: "specular_ior", ty: "float", default: Some(Val::float(1.5)) }, + InputDef { name: "specular_roughness_anisotropy", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "transmission_weight", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "transmission_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, + InputDef { name: "transmission_depth", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "transmission_scatter", ty: "color3", default: Some(Val::vec3(0.0, 0.0, 0.0)) }, + InputDef { name: "transmission_scatter_anisotropy", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "transmission_dispersion_scale", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "transmission_dispersion_abbe_number", ty: "float", default: Some(Val::float(20.0)) }, + InputDef { name: "subsurface_weight", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "subsurface_color", ty: "color3", default: Some(Val::vec3(0.8, 0.8, 0.8)) }, + InputDef { name: "subsurface_radius", ty: "float", default: Some(Val::float(1.0)) }, + InputDef { name: "subsurface_radius_scale", ty: "color3", default: Some(Val::vec3(1.0, 0.5, 0.25)) }, + InputDef { name: "subsurface_scatter_anisotropy", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "fuzz_weight", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "fuzz_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, + InputDef { name: "fuzz_roughness", ty: "float", default: Some(Val::float(0.5)) }, + InputDef { name: "coat_weight", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "coat_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, + InputDef { name: "coat_roughness", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "coat_roughness_anisotropy", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "coat_ior", ty: "float", default: Some(Val::float(1.6)) }, + InputDef { name: "coat_darkening", ty: "float", default: Some(Val::float(1.0)) }, + InputDef { name: "thin_film_weight", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "thin_film_thickness", ty: "float", default: Some(Val::float(0.5)) }, + InputDef { name: "thin_film_ior", ty: "float", default: Some(Val::float(1.4)) }, + InputDef { name: "emission_luminance", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "emission_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, + InputDef { name: "geometry_opacity", ty: "float", default: Some(Val::float(1.0)) }, + InputDef { name: "geometry_thin_walled", ty: "boolean", default: Some(Val::float(0.0)) }, + InputDef { name: "geometry_normal", ty: "vector3", default: None }, + InputDef { name: "geometry_coat_normal", ty: "vector3", default: None }, + InputDef { name: "geometry_tangent", ty: "vector3", default: None }, + InputDef { name: "geometry_coat_tangent", ty: "vector3", default: None }, + ]; + + /// `ND_standard_surface_surfaceshader`'s inputs, in nodedef order. + pub const STANDARD_SURFACE: &[InputDef] = &[ + InputDef { name: "base", ty: "float", default: Some(Val::float(1.0)) }, + InputDef { name: "base_color", ty: "color3", default: Some(Val::vec3(0.8, 0.8, 0.8)) }, + InputDef { name: "diffuse_roughness", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "metalness", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "specular", ty: "float", default: Some(Val::float(1.0)) }, + InputDef { name: "specular_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, + InputDef { name: "specular_roughness", ty: "float", default: Some(Val::float(0.2)) }, + InputDef { name: "specular_IOR", ty: "float", default: Some(Val::float(1.5)) }, + InputDef { name: "specular_anisotropy", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "specular_rotation", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "transmission", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "transmission_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, + InputDef { name: "transmission_depth", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "transmission_scatter", ty: "color3", default: Some(Val::vec3(0.0, 0.0, 0.0)) }, + InputDef { name: "transmission_scatter_anisotropy", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "transmission_dispersion", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "transmission_extra_roughness", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "subsurface", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "subsurface_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, + InputDef { name: "subsurface_radius", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, + InputDef { name: "subsurface_scale", ty: "float", default: Some(Val::float(1.0)) }, + InputDef { name: "subsurface_anisotropy", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "sheen", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "sheen_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, + InputDef { name: "sheen_roughness", ty: "float", default: Some(Val::float(0.3)) }, + InputDef { name: "coat", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "coat_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, + InputDef { name: "coat_roughness", ty: "float", default: Some(Val::float(0.1)) }, + InputDef { name: "coat_anisotropy", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "coat_rotation", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "coat_IOR", ty: "float", default: Some(Val::float(1.5)) }, + InputDef { name: "coat_normal", ty: "vector3", default: None }, + InputDef { name: "coat_affect_color", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "coat_affect_roughness", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "thin_film_thickness", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "thin_film_IOR", ty: "float", default: Some(Val::float(1.5)) }, + InputDef { name: "emission", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "emission_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, + InputDef { name: "opacity", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, + InputDef { name: "thin_walled", ty: "boolean", default: Some(Val::float(0.0)) }, + InputDef { name: "normal", ty: "vector3", default: None }, + InputDef { name: "tangent", ty: "vector3", default: None }, + ]; + + /// `ND_gltf_pbr_surfaceshader`'s inputs, in nodedef order. + pub const GLTF_PBR: &[InputDef] = &[ + InputDef { name: "base_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, + InputDef { name: "metallic", ty: "float", default: Some(Val::float(1.0)) }, + InputDef { name: "roughness", ty: "float", default: Some(Val::float(1.0)) }, + InputDef { name: "normal", ty: "vector3", default: None }, + InputDef { name: "tangent", ty: "vector3", default: None }, + InputDef { name: "occlusion", ty: "float", default: Some(Val::float(1.0)) }, + InputDef { name: "transmission", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "specular", ty: "float", default: Some(Val::float(1.0)) }, + InputDef { name: "specular_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, + InputDef { name: "ior", ty: "float", default: Some(Val::float(1.5)) }, + InputDef { name: "alpha", ty: "float", default: Some(Val::float(1.0)) }, + InputDef { name: "alpha_mode", ty: "integer", default: Some(Val::float(0.0)) }, + InputDef { name: "alpha_cutoff", ty: "float", default: Some(Val::float(0.5)) }, + InputDef { name: "iridescence", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "iridescence_ior", ty: "float", default: Some(Val::float(1.3)) }, + InputDef { name: "iridescence_thickness", ty: "float", default: Some(Val::float(100.0)) }, + InputDef { name: "sheen_color", ty: "color3", default: Some(Val::vec3(0.0, 0.0, 0.0)) }, + InputDef { name: "sheen_roughness", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "clearcoat", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "clearcoat_roughness", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "clearcoat_normal", ty: "vector3", default: None }, + InputDef { name: "emissive", ty: "color3", default: Some(Val::vec3(0.0, 0.0, 0.0)) }, + InputDef { name: "emissive_strength", ty: "float", default: Some(Val::float(1.0)) }, + InputDef { name: "thickness", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "attenuation_distance", ty: "float", default: None }, + InputDef { name: "attenuation_color", ty: "color3", default: Some(Val::vec3(1.0, 1.0, 1.0)) }, + InputDef { name: "anisotropy_strength", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "anisotropy_rotation", ty: "float", default: Some(Val::float(0.0)) }, + InputDef { name: "dispersion", ty: "float", default: Some(Val::float(0.0)) }, + ]; + +} +pub use tables::{GLTF_PBR, OPEN_PBR_SURFACE, STANDARD_SURFACE}; + +/// Expands `node` — an `open_pbr_surface`, `standard_surface` or `gltf_pbr` — +/// into `out`. +pub fn build(c: &mut Compiler<'_>, node: &Node, out: &mut Closures) { + let (defs, default_version): (&'static [InputDef], &str) = match node.category.as_str() { + "open_pbr_surface" => (OPEN_PBR_SURFACE, "1.1"), + "standard_surface" => (STANDARD_SURFACE, "1.0.1"), + "gltf_pbr" => (GLTF_PBR, "2.0.1"), + other => { + c.unsupported.insert(other.to_string()); + return; + } + }; + let mut b = B { + c, + node, + defs, + memo: HashMap::new(), + out, + }; + if let Some(v) = b.version() + && v != default_version + { + b.out.reported.insert(format!( + "{} version {v} (built as the default version {default_version})", + node.category + )); + } + match node.category.as_str() { + "open_pbr_surface" => open_pbr_surface(&mut b), + "standard_surface" => standard_surface(&mut b), + _ => gltf_pbr(&mut b), + } +} + +/// A builder over one surface node: its inputs by name, program ops, and the +/// tree under construction. +struct B<'x, 'a> { + c: &'x mut Compiler<'a>, + node: &'x Node, + defs: &'static [InputDef], + memo: HashMap<&'static str, Slot>, + out: &'x mut Closures, +} + +impl B<'_, '_> { + fn version(&self) -> Option<&str> { + self.node.version.as_deref() + } + + fn def(&self, name: &str) -> Option<&'static InputDef> { + self.defs.iter().find(|d| d.name == name) + } + + /// An input: its connection, its authored value, or its nodedef default. + fn get(&mut self, name: &'static str) -> Slot { + if let Some(&s) = self.memo.get(name) { + return s; + } + let default = self.def(name).and_then(|d| d.default).unwrap_or(Val::ZERO); + let s = self.c.input_or(self.node, name, default); + self.memo.insert(name, s); + s + } + + /// An input whose default is geometric (`Nworld` / `Tworld`): `None` + /// unless authored. + fn geom(&mut self, name: &'static str) -> Option { + self.c.optional_input(self.node, name) + } + + /// Reports `name` when it is authored away from its nodedef default. + fn report(&mut self, name: &'static str, why: &str) { + let default = self.def(name).and_then(|d| d.default).unwrap_or(Val::ZERO); + if crate::bsdf::authored_away(self.c, self.node, name, default) { + self.out.reported.insert(format!("{name} ({why})")); + } + } + + fn k(&mut self, x: f32) -> Slot { + self.c.constant(Val::float(x)) + } + + fn k3(&mut self, x: f32, y: f32, z: f32) -> Slot { + self.c.constant(Val::vec3(x, y, z)) + } + + fn bin(&mut self, op: BinOp, a: Slot, b: Slot) -> Slot { + self.c.emit(Op::Binary { op, a, b }) + } + + fn mul(&mut self, a: Slot, b: Slot) -> Slot { + self.bin(BinOp::Mul, a, b) + } + + fn add(&mut self, a: Slot, b: Slot) -> Slot { + self.bin(BinOp::Add, a, b) + } + + fn sub(&mut self, a: Slot, b: Slot) -> Slot { + self.bin(BinOp::Sub, a, b) + } + + fn div(&mut self, a: Slot, b: Slot) -> Slot { + self.bin(BinOp::Div, a, b) + } + + fn un(&mut self, op: UnOp, a: Slot) -> Slot { + self.c.emit(Op::Unary { op, a }) + } + + fn mix(&mut self, fg: Slot, bg: Slot, m: Slot) -> Slot { + self.c.emit(Op::Mix { fg, bg, m }) + } + + fn clamp(&mut self, a: Slot, lo: f32, hi: f32) -> Slot { + let (low, high) = (self.k(lo), self.k(hi)); + self.c.emit(Op::Clamp { a, low, high }) + } + + fn convert(&mut self, a: Slot, arity: u8) -> Slot { + self.c.emit(Op::Convert { a, arity }) + } + + fn extract(&mut self, a: Slot, index: usize) -> Slot { + self.c.emit(Op::Extract { a, index }) + } + + /// MaterialX's `ifgreater`: `in1` where `value1 > value2`, else `in2`. + /// + /// Built from existing operators so the JIT needs nothing new: + /// `m = max(sign(value1 − value2), 0)` is 1 exactly when `value1` is + /// greater — MaterialX's `sign(0)` is 0, and `x − x` is `+0`, which is + /// what makes equality pick `in2` — and `mix(fg = in1, bg = in2, m)` is + /// exact at `m ∈ {0, 1}` for finite operands (every divide and log in the + /// program is kept finite). + fn gt(&mut self, v1: Slot, v2: Slot, in1: Slot, in2: Slot) -> Slot { + let d = self.sub(v1, v2); + let s = self.un(UnOp::Sign, d); + let zero = self.k(0.0); + let m = self.bin(BinOp::Max, s, zero); + self.mix(in1, in2, m) + } + + /// MaterialX's `ifgreatereq`: `in1` where `value1 ≥ value2`, else `in2` — + /// [`B::gt`] with its operands swapped, which is exact for the same reason. + fn ge(&mut self, v1: Slot, v2: Slot, in1: Slot, in2: Slot) -> Slot { + self.gt(v2, v1, in2, in1) + } + + /// MaterialX's `ifequal`: `in1` where `value1 = value2`, else `in2`, + /// built as "neither is greater". + fn eq(&mut self, v1: Slot, v2: Slot, in1: Slot, in2: Slot) -> Slot { + let below = self.gt(v2, v1, in2, in1); + self.gt(v1, v2, in2, below) + } + + /// The right-handed angle, in radians, a graph's `rotate3d` of its tangent + /// by `degrees` about the normal turns it by, `None` when `degrees` folds + /// to 0. + /// + /// `mx_rotate_vector3` is `v·cos θ + (v × axis)·sin θ + axis·(axis·v)(1 − + /// cos θ)`, and `v × axis = −(axis × v)`: Rodrigues' formula at `−θ`. So + /// a `rotate3d` by `amount` degrees turns the tangent by `−amount` in the + /// right-handed sense [`Leaf::rotation`] is stated in. + fn tangent_rotation(&mut self, degrees: Slot) -> Option { + if self.is(degrees, 0.0) { + return None; + } + let k = self.k(-std::f32::consts::PI / 180.0); + Some(self.mul(degrees, k)) + } + + /// `1 − a`. + fn one_minus(&mut self, a: Slot) -> Slot { + let one = self.k(1.0); + self.c.emit(Op::Invert { a, amount: one }) + } + + /// Whether `s` folds to a constant whose every lane is `x`. + fn is(&self, s: Slot, x: f32) -> bool { + self.c.fold(s).is_some_and(|v| { + let lanes = (v.arity as usize).clamp(1, 4); + v.v[..lanes].iter().all(|&c| c == x) + }) + } + + // ---- the tree ---------------------------------------------------------- + + /// A leaf, pruned when its weight folds to zero. + fn leaf( + &mut self, + bsdf: Bsdf, + weight: Slot, + normal: Option, + tangent: Option, + ) -> Option { + if self.is(weight, 0.0) { + return None; + } + if let Bsdf::Sheen { + mode: SheenMode::Zeltner, + .. + } = &bsdf + { + self.out + .reported + .insert("sheen_bsdf mode zeltner (evaluated as conty_kulla)".into()); + } + Some(self.push(Closure::Leaf(Leaf { + bsdf, + weight, + normal, + tangent, + rotation: None, + }))) + } + + /// Turns the tangent of the leaf `id` by `rotation` ([`Leaf::rotation`]). + fn rotate(&mut self, id: Option, rotation: Option) { + if let (Some(id), Some(r)) = (id, rotation) + && let Closure::Leaf(l) = &mut self.out.nodes[id as usize] + { + l.rotation = Some(r); + } + } + + /// The surface's presence, unless it folds to 1. + fn set_opacity(&mut self, opacity: Slot) { + if !self.is(opacity, 1.0) { + self.out.opacity = Some(opacity); + } + } + + fn push(&mut self, c: Closure) -> NodeId { + self.out.nodes.push(c); + (self.out.nodes.len() - 1) as NodeId + } + + fn layer(&mut self, top: Option, base: Option) -> Option { + match (top, base) { + (Some(top), Some(base)) => Some(self.push(Closure::Layer { top, base })), + (t, b) => t.or(b), + } + } + + /// `mix(fg, bg, m)`, pruned at a constant `m` of 0 or 1. A pruned branch + /// keeps its share of the throughput ([`Closures::mix`]). + fn mixc(&mut self, fg: Option, bg: Option, m: Slot) -> Option { + if self.is(m, 0.0) { + return bg; + } + if self.is(m, 1.0) { + return fg; + } + self.out.mix(fg, bg, m) + } + + /// `multiply(x, w)`, pruned at a constant `w` of 0 and elided at 1. + fn multiply(&mut self, x: Option, w: Slot) -> Option { + let x = x?; + if self.is(w, 0.0) { + return None; + } + if self.is(w, 1.0) { + return Some(x); + } + Some(self.push(Closure::Multiply { + input: x, + weight: w, + })) + } + + /// A thin film, `None` when its thickness folds to zero. + fn film(&mut self, thickness_nm: Slot, ior: Slot) -> Option { + (!self.is(thickness_nm, 0.0)).then_some(ThinFilm { + thickness: thickness_nm, + ior, + }) + } + + /// An EDF term, pruned when its weight folds to zero. + fn emit_edf(&mut self, color: Slot, weight: Slot, falloff: Option) { + if self.is(weight, 0.0) || self.is(color, 0.0) { + return; + } + self.out.emission.push(Emission { + color, + weight, + falloff, + }); + } + + // ---- shared helper graphs ---------------------------------------------- + + /// `NG_open_pbr_anisotropy`: `α = r²`, `αx = α·√(2 / (1 + (1 − a)²))`, + /// `αy = (1 − a)·αx`. + fn open_pbr_anisotropy(&mut self, roughness: Slot, anisotropy: Slot) -> Slot { + let inv = self.one_minus(anisotropy); // aniso_invert + let inv_sq = self.mul(inv, inv); // aniso_invert_sq + let one = self.k(1.0); + let denom = self.add(inv_sq, one); // denom + let two = self.k(2.0); + let fraction = self.div(two, denom); // fraction + let sqrt = self.un(UnOp::Sqrt, fraction); // sqrt + let rough_sq = self.mul(roughness, roughness); // rough_sq + let ax = self.mul(rough_sq, sqrt); // alpha_x + let ay = self.mul(inv, ax); // alpha_y + self.c.emit(Op::Combine2 { a: ax, b: ay }) // result + } + + /// `ND_roughness_anisotropy` (`mx_roughness_anisotropy.glsl`): + /// `α = clamp(r², ε, 1)`; with `aspect = √(1 − clamp(a, 0, 0.98))`, + /// `(min(α / aspect, 1), α · aspect)` — which is `(α, α)` at `a ≤ 0`, + /// so the GLSL's branch needs no select here. + fn roughness_anisotropy(&mut self, roughness: Slot, anisotropy: Slot) -> Slot { + let r2 = self.mul(roughness, roughness); + let alpha = self.clamp(r2, 1e-8, 1.0); + let a = self.clamp(anisotropy, 0.0, 0.98); + let one_minus = self.one_minus(a); + let aspect = self.un(UnOp::Sqrt, one_minus); + let x0 = self.div(alpha, aspect); + let one = self.k(1.0); + let x = self.bin(BinOp::Min, x0, one); + let y = self.mul(alpha, aspect); + self.c.emit(Op::Combine2 { a: x, b: y }) + } + + /// `((ior − 1) / (ior + 1))²`. + fn ior_to_f0(&mut self, ior: Slot) -> Slot { + let one = self.k(1.0); + let m = self.sub(ior, one); + let p = self.add(one, ior); + let r = self.div(m, p); + self.mul(r, r) + } +} diff --git a/crates/crust-mtlx/src/surface/open_pbr.rs b/crates/crust-mtlx/src/surface/open_pbr.rs new file mode 100644 index 00000000..25146f10 --- /dev/null +++ b/crates/crust-mtlx/src/surface/open_pbr.rs @@ -0,0 +1,332 @@ +//! open_pbr_surface — NG_open_pbr_surface_surfaceshader (OpenPBR 1.1) + +use super::*; + +pub(super) fn open_pbr_surface(b: &mut B<'_, '_>) { + b.report( + "transmission_dispersion_scale", + "ignored, as MaterialX's own graph does", + ); + + let normal = b.geom("geometry_normal"); + let tangent = b.geom("geometry_tangent"); + let coat_normal = b.geom("geometry_coat_normal"); + let coat_tangent = b.geom("geometry_coat_tangent"); + + // Roughness: the coat broadens the base specular. + let coat_roughness = b.get("coat_roughness"); + let specular_roughness = b.get("specular_roughness"); + let coat_weight = b.get("coat_weight"); + let four = b.k(4.0); + let cr4 = b.bin(BinOp::Pow, coat_roughness, four); // coat_roughness_to_power_4 + let two = b.k(2.0); + let two_cr4 = b.mul(cr4, two); // two_times_coat_roughness_to_power_4 + let sr4 = b.bin(BinOp::Pow, specular_roughness, four); // specular_roughness_to_power_4 + let sum = b.add(two_cr4, sr4); // add_coat_and_spec_roughnesses_to_power_4 + let one = b.k(1.0); + let min1 = b.bin(BinOp::Min, one, sum); // min_1_add_coat_and_spec_roughnesses_to_power_4 + let quarter = b.k(0.25); + let coat_affected = b.bin(BinOp::Pow, min1, quarter); // coat_affected_specular_roughness + let effective = b.mix(coat_affected, specular_roughness, coat_weight); // effective_specular_roughness + let aniso = b.get("specular_roughness_anisotropy"); + let main_roughness = b.open_pbr_anisotropy(effective, aniso); // main_roughness + + // Subsurface, thin-walled and not, built only when it is live: a + // literal-zero `subsurface_weight` (the default) makes `opaque_base` the + // diffuse alone. + let subsurface_color = b.get("subsurface_color"); + let zero = b.k(0.0); + let diffuse_roughness = b.get("base_diffuse_roughness"); + let one_w = b.k(1.0); + let base_color = b.get("base_color"); + let bcn = b.bin(BinOp::Max, base_color, zero); // base_color_nonnegative + let base_weight = b.get("base_weight"); + let diffuse = b.leaf( + Bsdf::Diffuse { + model: DiffuseModel::Eon, + color: bcn, + roughness: diffuse_roughness, + }, + base_weight, + normal, + None, + ); // diffuse_bsdf + let thin_walled = b.get("geometry_thin_walled"); + let subsurface_weight = b.get("subsurface_weight"); + let opaque_base = if b.is(subsurface_weight, 0.0) { + diffuse + } else { + let ssc = b.bin(BinOp::Max, subsurface_color, zero); // subsurface_color_nonnegative + let sss_refl_bsdf = b.leaf( + Bsdf::Diffuse { + model: DiffuseModel::OrenNayar, + color: ssc, + roughness: diffuse_roughness, + }, + one_w, + normal, + None, + ); // subsurface_thin_walled_reflection_bsdf + let ss_aniso = b.get("subsurface_scatter_anisotropy"); + let one_minus_aniso = b.one_minus(ss_aniso); // one_minus_subsurface_scatter_anisotropy + let brdf_factor = b.mul(subsurface_color, one_minus_aniso); // subsurface_thin_walled_brdf_factor + let sss_refl = b.multiply(sss_refl_bsdf, brdf_factor); // subsurface_thin_walled_reflection + let sss_trans_bsdf = b.leaf(Bsdf::Translucent { color: ssc }, one_w, normal, None); // subsurface_thin_walled_transmission_bsdf + let one = b.k(1.0); + let one_plus_aniso = b.add(one, ss_aniso); // one_plus_subsurface_scatter_anisotropy + let btdf_factor = b.mul(subsurface_color, one_plus_aniso); // subsurface_thin_walled_btdf_factor + let sss_trans = b.multiply(sss_trans_bsdf, btdf_factor); // subsurface_thin_walled_transmission + let half = b.k(0.5); + let sss_thin = b.mixc(sss_refl, sss_trans, half); // subsurface_thin_walled + let radius_scale = b.get("subsurface_radius_scale"); + let radius = b.get("subsurface_radius"); + let radius_scaled = b.mul(radius_scale, radius); // subsurface_radius_scaled + let sss_bsdf = b.leaf( + Bsdf::Subsurface { + color: ssc, + radius: radius_scaled, + anisotropy: ss_aniso, + }, + one_w, + normal, + None, + ); // subsurface_bsdf + let selector = b.convert(thin_walled, 1); // subsurface_selector + let selected = b.mixc(sss_thin, sss_bsdf, selector); // selected_subsurface + b.mixc(selected, diffuse, subsurface_weight) // opaque_base + }; + + // The transmission volume. + let transmission_color = b.get("transmission_color"); + let tcv = b.convert(transmission_color, 3); // transmission_color_vector + let tcl = b.un(UnOp::Ln, tcv); // transmission_color_ln + let minus_one = b.k(-1.0); + let ext_den = b.mul(tcl, minus_one); // extinction_coeff_denom + let depth = b.get("transmission_depth"); + let depth_v = b.convert(depth, 3); // transmission_depth_vector + let extinction = b.div(ext_den, depth_v); // extinction_coeff + let scatter = b.get("transmission_scatter"); + let scatter_v = b.convert(scatter, 3); // transmission_scatter_vector + let scattering = b.div(scatter_v, depth_v); // scattering_coeff + let absorption = b.sub(extinction, scattering); // absorption_coeff + let ax = b.extract(absorption, 0); + let ay = b.extract(absorption, 1); + let az = b.extract(absorption, 2); + let min_xy = b.bin(BinOp::Min, ax, ay); + let amin = b.bin(BinOp::Min, min_xy, az); // absorption_coeff_min + let amin_v = b.convert(amin, 3); + let shifted = b.sub(absorption, amin_v); // absorption_coeff_shifted + let if_shifted = b.gt(zero, amin, shifted, absorption); // if_absorption_coeff_shifted + let zero3 = b.k3(0.0, 0.0, 0.0); + let vol_abs = b.gt(depth, zero, if_shifted, zero3); // if_volume_absorption + let vol_scat = b.gt(depth, zero, scattering, zero3); // if_volume_scattering + let vol_aniso = b.get("transmission_scatter_anisotropy"); + let transmission_weight = b.get("transmission_weight"); + let one = b.k(1.0); + if !b.is(transmission_weight, 0.0) { + b.out.volume = Some(Volume { + absorption: vol_abs, + scattering: vol_scat, + anisotropy: vol_aniso, + }); // dielectric_volume + } + + // The dielectric interface's IOR: relative to the coat, and modulated by + // specular_weight through F0. + let tf_thickness = b.get("thin_film_thickness"); + let thousand = b.k(1000.0); + let tf_nm = b.mul(tf_thickness, thousand); // thin_film_thickness_nm + let specular_ior = b.get("specular_ior"); + let coat_ior = b.get("coat_ior"); + let s2c = b.div(specular_ior, coat_ior); // specular_to_coat_ior_ratio + let c2s = b.div(coat_ior, specular_ior); // coat_to_specular_ior_ratio + let tir_fix = b.gt(s2c, one, s2c, c2s); // specular_to_coat_ior_ratio_tir_fix + let eta_s = b.mix(tir_fix, specular_ior, coat_weight); // eta_s + let em1 = b.sub(eta_s, one); // eta_s_minus_one + let ep1 = b.add(eta_s, one); // eta_s_plus_one + let f0_sqrt = b.div(em1, ep1); // specular_F0_sqrt + let f0 = b.mul(f0_sqrt, f0_sqrt); // specular_F0 + let specular_weight = b.get("specular_weight"); + let scaled_f0 = b.mul(specular_weight, f0); // scaled_specular_F0 + let scaled_f0c = b.clamp(scaled_f0, 0.0, 0.99999); // scaled_specular_F0_clamped + let sqrt_f0 = b.un(UnOp::Sqrt, scaled_f0c); // sqrt_scaled_specular_F0 + let sign = b.un(UnOp::Sign, em1); // sign_eta_s_minus_one + let eps = b.mul(sign, sqrt_f0); // modulated_eta_s_epsilon + let one_minus_eps = b.one_minus(eps); // one_minus_modulated_eta_s_epsilon + let one_plus_eps = b.add(one, eps); // one_plus_modulated_eta_s_epsilon + let modulated_eta = b.div(one_plus_eps, one_minus_eps); // modulated_eta_s + + // Transmission over the opaque base, then the reflection over that. + let white = b.k3(1.0, 1.0, 1.0); + let substrate = if b.is(transmission_weight, 0.0) { + opaque_base + } else { + let t_tint = b.gt(depth, zero, white, transmission_color); // if_transmission_tint + let d_trans = b.leaf( + Bsdf::Dielectric { + tint: t_tint, + ior: modulated_eta, + roughness: main_roughness, + mode: ScatterMode::T, + thin_film: None, + abbe: None, + }, + one_w, + normal, + tangent, + ); // dielectric_transmission (+ dielectric_volume_transmission) + b.mixc(d_trans, opaque_base, transmission_weight) // dielectric_substrate + }; + let specular_color = b.get("specular_color"); + let d_refl = b.leaf( + Bsdf::Dielectric { + tint: specular_color, + ior: modulated_eta, + roughness: main_roughness, + mode: ScatterMode::R, + thin_film: None, + abbe: None, + }, + one_w, + normal, + tangent, + ); // dielectric_reflection + let tf_ior = b.get("thin_film_ior"); + let thin_film_weight = b.get("thin_film_weight"); + let d_refl_mix = if b.is(thin_film_weight, 0.0) { + d_refl + } else { + let film = b.film(tf_nm, tf_ior); + let d_refl_tf = b.leaf( + Bsdf::Dielectric { + tint: specular_color, + ior: modulated_eta, + roughness: main_roughness, + mode: ScatterMode::R, + thin_film: film, + abbe: None, + }, + one_w, + normal, + tangent, + ); // dielectric_reflection_tf + b.mixc(d_refl_tf, d_refl, thin_film_weight) // dielectric_reflection_tf_mix + }; + let dielectric_base = b.layer(d_refl_mix, substrate); // dielectric_base + + // The metal. + let metal_refl = b.mul(base_color, base_weight); // metal_reflectivity + let metal_edge = b.mul(specular_color, specular_weight); // metal_edgecolor + let five = b.k(5.0); + let metalness = b.get("base_metalness"); + let metal_mix = if b.is(metalness, 0.0) { + None + } else { + let metal = b.leaf( + Bsdf::Schlick { + color0: metal_refl, + color82: metal_edge, + color90: white, + exponent: five, + roughness: main_roughness, + mode: ScatterMode::R, + thin_film: None, + }, + specular_weight, + normal, + tangent, + ); // metal_bsdf + if b.is(thin_film_weight, 0.0) { + metal + } else { + let film = b.film(tf_nm, tf_ior); + let metal_tf = b.leaf( + Bsdf::Schlick { + color0: metal_refl, + color82: metal_edge, + color90: white, + exponent: five, + roughness: main_roughness, + mode: ScatterMode::R, + thin_film: film, + }, + specular_weight, + normal, + tangent, + ); // metal_bsdf_tf + b.mixc(metal_tf, metal, thin_film_weight) // metal_bsdf_tf_mix + } + }; + let base_substrate = b.mixc(metal_mix, dielectric_base, metalness); // base_substrate + + // Coat darkening and tint. + let coat_f0 = b.ior_to_f0(coat_ior); // coat_ior_to_F0 + let one_minus_coat_f0 = b.one_minus(coat_f0); // one_minus_coat_F0 + let coat_ior_sq = b.mul(coat_ior, coat_ior); // coat_ior_sqr + let omf0_eta2 = b.div(one_minus_coat_f0, coat_ior_sq); // one_minus_coat_F0_over_eta2 + let k_coat = b.one_minus(omf0_eta2); // Kcoat + let e_metal = b.mul(base_color, specular_weight); // Emetal + let e_diel = b.mix(subsurface_color, base_color, subsurface_weight); // Edielectric + let e_base = b.mix(e_metal, e_diel, metalness); // Ebase + let ebk = b.mul(e_base, k_coat); // Ebase_Kcoat + let one_minus_k = b.one_minus(k_coat); // one_minus_Kcoat + let one_minus_ebk = b.sub(white, ebk); // one_minus_Ebase_Kcoat + let omk3 = b.convert(one_minus_k, 3); // one_minus_Kcoat_color + let darkening = b.div(omk3, one_minus_ebk); // base_darkening + let coat_darkening = b.get("coat_darkening"); + let cwd = b.mul(coat_weight, coat_darkening); // coat_weight_times_coat_darkening + let mod_dark = b.mix(darkening, white, cwd); // modulated_base_darkening + let darkened = b.multiply(base_substrate, mod_dark); // darkened_base_substrate + let coat_color = b.get("coat_color"); + let coat_att = b.mix(coat_color, white, coat_weight); // coat_attenuation + let attenuated = b.multiply(darkened, coat_att); // coat_substrate_attenuated + let coat_aniso = b.get("coat_roughness_anisotropy"); + let coat_rough_v = b.open_pbr_anisotropy(coat_roughness, coat_aniso); // coat_roughness_vector + let coat = b.leaf( + Bsdf::Dielectric { + tint: white, + ior: coat_ior, + roughness: coat_rough_v, + mode: ScatterMode::R, + thin_film: None, + abbe: None, + }, + coat_weight, + coat_normal, + coat_tangent, + ); // coat_bsdf + let coat_layer = b.layer(coat, attenuated); // coat_layer + let fuzz_color = b.get("fuzz_color"); + let fuzz_roughness = b.get("fuzz_roughness"); + let fuzz_weight = b.get("fuzz_weight"); + let fuzz = b.leaf( + Bsdf::Sheen { + color: fuzz_color, + roughness: fuzz_roughness, + mode: SheenMode::Zeltner, + }, + fuzz_weight, + normal, + None, + ); // fuzz_bsdf + b.out.root = b.layer(fuzz, coat_layer); // fuzz_layer + + // Emission: uncoated, and through the coat's Fresnel. + let emission_color = b.get("emission_color"); + let luminance = b.get("emission_luminance"); + let ew = b.mul(emission_color, luminance); // emission_weight + let uncoated_w = b.one_minus(coat_weight); + b.emit_edf(ew, uncoated_w, None); // uncoated_emission_edf (bg of emission_edf) + let coated_w = b.mul(coat_color, coat_weight); // coat_tinted_emission_edf · mix + let c0 = b.convert(one_minus_coat_f0, 3); // one_minus_coat_F0_color + let falloff = EdfFalloff { + color0: c0, + color90: zero3, + exponent: five, + }; + b.emit_edf(ew, coated_w, Some(falloff)); // coated_emission_edf (fg) + + b.out.thin_walled = Some(thin_walled); + let opacity = b.get("geometry_opacity"); + b.set_opacity(opacity); // shader_constructor.opacity +} diff --git a/crates/crust-mtlx/src/surface/standard_surface.rs b/crates/crust-mtlx/src/surface/standard_surface.rs new file mode 100644 index 00000000..9f636b72 --- /dev/null +++ b/crates/crust-mtlx/src/surface/standard_surface.rs @@ -0,0 +1,251 @@ +//! standard_surface — NG_standard_surface_surfaceshader_100 + +use super::*; + +pub(super) fn standard_surface(b: &mut B<'_, '_>) { + for (name, why) in [ + ( + "transmission_depth", + "ignored, as MaterialX's own graph does", + ), + ( + "transmission_scatter", + "ignored, as MaterialX's own graph does", + ), + ( + "transmission_dispersion", + "ignored, as MaterialX's own graph does", + ), + ] { + b.report(name, why); + } + + let normal = b.geom("normal"); + let coat_normal = b.geom("coat_normal"); + let tangent = b.geom("tangent"); + // `main_tangent` / `coat_tangent`: the tangent turned by a fraction of a + // full turn about `normal` / `coat_normal`, only where the lobe is + // anisotropic — the leaves' own normals, so a leaf rotation. + let main_rotation = standard_rotation(b, "specular_rotation", "specular_anisotropy"); + let coat_rotation = standard_rotation(b, "coat_rotation", "coat_anisotropy"); + + let coat_affect_roughness = b.get("coat_affect_roughness"); + let coat = b.get("coat"); + let coat_roughness = b.get("coat_roughness"); + let m1 = b.mul(coat_affect_roughness, coat); // coat_affect_roughness_multiply1 + let m2 = b.mul(m1, coat_roughness); // coat_affect_roughness_multiply2 + let specular_roughness = b.get("specular_roughness"); + let one = b.k(1.0); + let coat_affected = b.mix(one, specular_roughness, m2); // coat_affected_roughness + let specular_anisotropy = b.get("specular_anisotropy"); + let main_roughness = b.roughness_anisotropy(coat_affected, specular_anisotropy); // main_roughness + let extra = b.get("transmission_extra_roughness"); + let tr_add = b.add(specular_roughness, extra); // transmission_roughness_add + let tr_clamped = b.clamp(tr_add, 0.0, 1.0); // transmission_roughness_clamped + let coat_affected_tr = b.mix(one, tr_clamped, m2); // coat_affected_transmission_roughness + let transmission_roughness = b.roughness_anisotropy(coat_affected_tr, specular_anisotropy); // transmission_roughness + + let coat_clamped = b.clamp(coat, 0.0, 1.0); // coat_clamped + let coat_affect_color = b.get("coat_affect_color"); + let cg_m = b.mul(coat_clamped, coat_affect_color); // coat_gamma_multiply + let coat_gamma = b.add(cg_m, one); // coat_gamma + let zero = b.k(0.0); + let base_color = b.get("base_color"); + let bcn = b.bin(BinOp::Max, base_color, zero); // base_color_nonnegative + let diffuse_color = b.bin(BinOp::Pow, bcn, coat_gamma); // coat_affected_diffuse_color + let subsurface_color = b.get("subsurface_color"); + let scn = b.bin(BinOp::Max, subsurface_color, zero); // subsurface_color_nonnegative + let sss_color = b.bin(BinOp::Pow, scn, coat_gamma); // coat_affected_subsurface_color + + let base = b.get("base"); + let diffuse_roughness = b.get("diffuse_roughness"); + let diffuse = b.leaf( + Bsdf::Diffuse { + model: DiffuseModel::OrenNayar, + color: diffuse_color, + roughness: diffuse_roughness, + }, + base, + normal, + None, + ); // diffuse_bsdf + let subsurface = b.get("subsurface"); + let sss_mix = if b.is(subsurface, 0.0) { + diffuse + } else { + let one_w = b.k(1.0); + let translucent = b.leaf(Bsdf::Translucent { color: sss_color }, one_w, normal, None); // translucent_bsdf + let radius = b.get("subsurface_radius"); + let scale = b.get("subsurface_scale"); + let radius_scaled = b.mul(radius, scale); // subsurface_radius_scaled + let ss_aniso = b.get("subsurface_anisotropy"); + let sss = b.leaf( + Bsdf::Subsurface { + color: sss_color, + radius: radius_scaled, + anisotropy: ss_aniso, + }, + one_w, + normal, + None, + ); // subsurface_bsdf + let thin_walled = b.get("thin_walled"); + let selector = b.convert(thin_walled, 1); // subsurface_selector + let selected = b.mixc(translucent, sss, selector); // selected_subsurface_bsdf + b.mixc(selected, diffuse, subsurface) // subsurface_mix + }; + let sheen_w = b.get("sheen"); + let sheen_color = b.get("sheen_color"); + let sheen_roughness = b.get("sheen_roughness"); + let sheen = b.leaf( + Bsdf::Sheen { + color: sheen_color, + roughness: sheen_roughness, + mode: SheenMode::ContyKulla, + }, + sheen_w, + normal, + None, + ); // sheen_bsdf + let sheen_layer = b.layer(sheen, sss_mix); // sheen_layer + let transmission = b.get("transmission"); + let transmission_color = b.get("transmission_color"); + let specular_ior = b.get("specular_IOR"); + let one_w = b.k(1.0); + let trans_mix = if b.is(transmission, 0.0) { + sheen_layer + } else { + let trans = b.leaf( + Bsdf::Dielectric { + tint: transmission_color, + ior: specular_ior, + roughness: transmission_roughness, + mode: ScatterMode::T, + thin_film: None, + abbe: None, + }, + one_w, + normal, + tangent, + ); // transmission_bsdf + b.rotate(trans, main_rotation); + b.mixc(trans, sheen_layer, transmission) // transmission_mix + }; + let specular = b.get("specular"); + let specular_color = b.get("specular_color"); + let tf_thickness = b.get("thin_film_thickness"); + let tf_ior = b.get("thin_film_IOR"); + let film = b.film(tf_thickness, tf_ior); + let spec = b.leaf( + Bsdf::Dielectric { + tint: specular_color, + ior: specular_ior, + roughness: main_roughness, + mode: ScatterMode::R, + thin_film: film, + abbe: None, + }, + specular, + normal, + tangent, + ); // specular_bsdf + b.rotate(spec, main_rotation); + let spec_layer = b.layer(spec, trans_mix); // specular_layer + let metalness = b.get("metalness"); + let metal_mix = if b.is(metalness, 0.0) { + spec_layer + } else { + let metal_refl = b.mul(base_color, base); // metal_reflectivity + let metal_edge = b.mul(specular_color, specular); // metal_edgecolor + let n = b.c.emit(Op::ArtisticIor { + reflectivity: metal_refl, + edge: metal_edge, + extinction: false, + }); // artistic_ior.ior + let k = b.c.emit(Op::ArtisticIor { + reflectivity: metal_refl, + edge: metal_edge, + extinction: true, + }); // artistic_ior.extinction + let metal = b.leaf( + Bsdf::Conductor { + ior: n, + extinction: k, + roughness: main_roughness, + thin_film: film, + }, + one_w, + normal, + tangent, + ); // metal_bsdf + b.rotate(metal, main_rotation); + b.mixc(metal, spec_layer, metalness) // metalness_mix + }; + let coat_color = b.get("coat_color"); + let white = b.k3(1.0, 1.0, 1.0); + let coat_att = b.mix(coat_color, white, coat); // coat_attenuation + let attenuated = b.multiply(metal_mix, coat_att); // thin_film_layer_attenuated + let coat_anisotropy = b.get("coat_anisotropy"); + let coat_rough_v = b.roughness_anisotropy(coat_roughness, coat_anisotropy); // coat_roughness_vector + let coat_ior = b.get("coat_IOR"); + let coat_bsdf = b.leaf( + Bsdf::Dielectric { + tint: white, + ior: coat_ior, + roughness: coat_rough_v, + mode: ScatterMode::R, + thin_film: None, + abbe: None, + }, + coat, + coat_normal, + tangent, + ); // coat_bsdf + b.rotate(coat_bsdf, coat_rotation); + b.out.root = b.layer(coat_bsdf, attenuated); // coat_layer + + // Emission, uncoated and through the coat's Fresnel. + let coat_f0 = b.ior_to_f0(coat_ior); // coat_ior_to_F0 + let one_minus_f0 = b.one_minus(coat_f0); // one_minus_coat_ior_to_F0 + let emission_color = b.get("emission_color"); + let emission = b.get("emission"); + let ew = b.mul(emission_color, emission); // emission_weight + let uncoated_w = b.one_minus(coat); + b.emit_edf(ew, uncoated_w, None); // emission_edf (bg of blended_coat_emission_edf) + let coated_w = b.mul(coat_color, coat); // coat_tinted_emission_edf · mix + let c0 = b.convert(one_minus_f0, 3); // emission_color0 + let zero3 = b.k3(0.0, 0.0, 0.0); + let five = b.k(5.0); + b.emit_edf( + ew, + coated_w, + Some(EdfFalloff { + color0: c0, + color90: zero3, + exponent: five, + }), + ); // coat_emission_edf (fg) + + let thin_walled = b.get("thin_walled"); + b.out.thin_walled = Some(thin_walled); + let opacity = b.get("opacity"); + let luminance = b.c.luminance(opacity); // opacity_luminance(_float) + b.set_opacity(luminance); // shader_constructor.opacity +} + +/// `standard_surface`'s `main_tangent` / `coat_tangent`: `rotate3d` by +/// `rotation · 360` degrees, selected by `ifgreater(anisotropy, 0)`. +fn standard_rotation( + b: &mut B<'_, '_>, + rotation: &'static str, + anisotropy: &'static str, +) -> Option { + let r = b.get(rotation); + let full_turn = b.k(360.0); + let degrees = b.mul(r, full_turn); // tangent_rotate_degree + let angle = b.tangent_rotation(degrees)?; // tangent_rotate + let a = b.get(anisotropy); + let zero = b.k(0.0); + let selected = b.gt(a, zero, angle, zero); // main_tangent / coat_tangent + (!b.is(selected, 0.0)).then_some(selected) +} diff --git a/crates/crust-render/src/logging.rs b/crates/crust-render/src/logging.rs new file mode 100644 index 00000000..b4a9b1e8 --- /dev/null +++ b/crates/crust-render/src/logging.rs @@ -0,0 +1,268 @@ +//! Logging: the `tracing` subscriber `-l` and `--log-file` configure, and the +//! one target (`--stats`) no level suppresses. + +use std::path::Path; +use std::sync::Mutex; +use std::time::SystemTime; + +use tracing::Level; +use tracing_subscriber::filter::filter_fn; +use tracing_subscriber::fmt; +use tracing_subscriber::layer::SubscriberExt; +use tracing_subscriber::util::SubscriberInitExt; + +#[derive(clap::ValueEnum, Clone, Debug, Copy)] +pub(crate) enum LoggerLevel { + Debug, + Info, + Warn, + Error, + Trace, +} + +/// Target the `--stats` report is emitted under. +/// +/// It exists so the report can be exempted from `-l`: `--stats` is an +/// explicit request for the report, and honouring it only at `-l info` or +/// below would mean `--stats -l warn` silently produced nothing. The filter +/// in [`init`] admits this target at any level and applies `-l` to everything +/// else, which is what keeps the report a log event — reaching `--log-file` +/// like any other — without letting the log level decide whether it appears. +pub(crate) const STATS_TARGET: &str = "crust_render::stats"; + +/// Whether an event at `level` on `target` survives a `-l max` filter. +/// +/// Named rather than inlined into the closure so it can be tested: the whole +/// point of it is the one case that is easy to regress into silence — +/// [`STATS_TARGET`] passing at a level that rejects everything else. +fn event_enabled(target: &str, level: &Level, max: Level) -> bool { + target == STATS_TARGET || effective_level(target, level) <= max +} + +/// The level an event is filtered at, which for a few dependencies is not +/// the level it was emitted at. +/// +/// `cranelift_jit` logs the whole IR of every function it defines at INFO — +/// one multi-hundred-line dump per MaterialX program, so a default render's +/// INFO output grew with the number of materials, against the rule that INFO +/// lines do not scale with the scene. `tracing` cannot rewrite an event's +/// level, so it is *filtered* as DEBUG (shown from `-l debug` on) while still +/// printing its own `INFO` stamp. WARN and ERROR from cranelift are untouched. +fn effective_level(target: &str, level: &Level) -> Level { + if *level == Level::INFO && target.starts_with("cranelift") { + Level::DEBUG + } else { + *level + } +} + +pub(crate) fn get_logger_level(level: LoggerLevel) -> Level { + match level { + LoggerLevel::Debug => Level::DEBUG, + LoggerLevel::Info => Level::INFO, + LoggerLevel::Warn => Level::WARN, + LoggerLevel::Error => Level::ERROR, + LoggerLevel::Trace => Level::TRACE, + } +} + +/// Installs the process's subscriber: the terminal, and `--log-file`'s file +/// when one was asked for. The error is the message to print. +/// +/// Two layers rather than one writer teed into both, because ANSI is a +/// per-layer setting: a single writer would either colour the file with escape +/// codes or strip the colour from the terminal. The registry that composes them +/// costs no new dependency — `sharded-slab` and `thread_local` are already in +/// the graph via the `fmt` feature. +/// +/// The file is written unbuffered, deliberately: the subscriber that owns it is +/// the process-global one, which is never dropped, so a `BufWriter` would never +/// be flushed and would lose exactly the last lines — the ones explaining why a +/// run stopped. A log at these volumes is not worth a flush guard. +pub(crate) fn init(level: LoggerLevel, log_dir: Option<&Path>) -> Result<(), String> { + let log_file = log_dir.map(open_log_file).transpose()?; + let level = get_logger_level(level); + tracing_subscriber::registry() + // `-l` for everything except the `--stats` report, which the user + // asked for by flag and which therefore is not the log level's to + // suppress. See `STATS_TARGET`. + .with(filter_fn(move |meta| { + event_enabled(meta.target(), meta.level(), level) + })) + .with(fmt::layer()) + .with(log_file.map(|f| fmt::layer().with_ansi(false).with_writer(Mutex::new(f)))) + .init(); + Ok(()) +} + +/// A filename-safe UTC timestamp, `YYYYMMDDTHHMMSSZ`. +/// +/// Hand-rolled rather than pulled from `chrono` or `time`: neither is in the +/// dependency graph, and adding one to name a file would be the largest +/// dependency in this binary. `tracing-subscriber` formats its own line +/// timestamps the same way and for the same reason, so the `Z` suffix here +/// matches what the log lines themselves carry. +/// +/// The civil-from-days conversion is Howard Hinnant's, shifting the era to +/// start on 0000-03-01 so a leap day lands at the end of a 400-year cycle and +/// the month arithmetic needs no table. Valid for any date this can be handed. +pub(crate) fn utc_stamp(t: std::time::SystemTime) -> String { + let secs = t + .duration_since(std::time::UNIX_EPOCH) + .map(|d| d.as_secs() as i64) + // A clock before 1970 is not worth a failure path; it only names a file. + .unwrap_or(0); + let (days, rem) = (secs.div_euclid(86_400), secs.rem_euclid(86_400)); + let (hour, min, sec) = (rem / 3600, (rem % 3600) / 60, rem % 60); + + // Days since 1970-01-01 -> civil date. + let z = days + 719_468; + let era = z.div_euclid(146_097); + let doe = z.rem_euclid(146_097); + let yoe = (doe - doe / 1460 + doe / 36_524 - doe / 146_096) / 365; + let doy = doe - (365 * yoe + yoe / 4 - yoe / 100); + let mp = (5 * doy + 2) / 153; + let day = doy - (153 * mp + 2) / 5 + 1; + // `mp` counts from March; roll it back to a calendar month, and with it + // the year, which only advances once January is reached. + let month = if mp < 10 { mp + 3 } else { mp - 9 }; + let year = era * 400 + yoe + i64::from(month <= 2); + + format!("{year:04}{month:02}{day:02}T{hour:02}{min:02}{sec:02}Z") +} + +/// Opens the run's log file, creating any missing directories in `dir`. +/// +/// A failure fails the run rather than warning: nothing has been rendered yet +/// when this runs, so stopping costs no work, and a `--log-file` that quietly +/// produced no file would be discovered only after the render it was meant to +/// record. The error is the message to print. +fn open_log_file(dir: &Path) -> std::result::Result { + let path = dir.join(format!("crust-render-{}.log", utc_stamp(SystemTime::now()))); + if let Some(parent) = path.parent() + && !parent.as_os_str().is_empty() + && let Err(e) = std::fs::create_dir_all(parent) + { + return Err(format!( + "could not create log directory {}: {e}", + parent.display() + )); + } + let f = std::fs::File::create(&path) + .map_err(|e| format!("could not create log file {}: {e}", path.display()))?; + // Said on stderr rather than through `tracing`: the subscriber this file + // belongs to does not exist yet. + eprintln!("Logging to {}", path.display()); + Ok(f) +} + +#[cfg(test)] +mod tests { + use super::*; + + /// `SystemTime` at a given Unix second, for pinning `utc_stamp` against + /// dates whose answers are known independently. + fn at(unix_secs: u64) -> SystemTime { + std::time::UNIX_EPOCH + std::time::Duration::from_secs(unix_secs) + } + + #[test] + fn utc_stamp_names_known_instants() { + assert_eq!(utc_stamp(at(0)), "19700101T000000Z"); + assert_eq!(utc_stamp(at(1_774_267_884)), "20260323T121124Z"); + // Last second of a year, and the first of the next. + assert_eq!(utc_stamp(at(1_767_225_599)), "20251231T235959Z"); + assert_eq!(utc_stamp(at(1_767_225_600)), "20260101T000000Z"); + } + + #[test] + fn utc_stamp_handles_leap_years() { + // 2024 is a leap year: Feb 29 exists. + assert_eq!(utc_stamp(at(1_709_164_800)), "20240229T000000Z"); + // 2000 is a leap year (divisible by 400) — the case a naive + // "divisible by 4, except by 100" rule gets wrong. + assert_eq!(utc_stamp(at(951_782_400)), "20000229T000000Z"); + // 1900 was NOT a leap year, but it predates the epoch, so check the + // other end of the same rule: 2100 is not one either, and March 1 + // must follow February 28. + assert_eq!(utc_stamp(at(4_107_456_000)), "21000228T000000Z"); + assert_eq!(utc_stamp(at(4_107_542_400)), "21000301T000000Z"); + } + + #[test] + fn utc_stamp_is_filename_safe_and_sorts_chronologically() { + let mut prev = utc_stamp(at(0)); + for day in 1..4000u64 { + // Every 37 days, so the walk crosses month and year boundaries + // at varied offsets rather than landing on the same day each time. + let t = utc_stamp(at(day * 37 * 86_400 + 3661)); + assert!( + t.chars().all(|c| c.is_ascii_alphanumeric()), + "{t} is not filename-safe" + ); + assert_eq!(t.len(), 16, "{t} is not a fixed-width stamp"); + // Fixed width and zero-padded, so lexical order is chronological + // — which is the whole reason for this format over a locale one. + assert!(t > prev, "{t} does not sort after {prev}"); + prev = t; + } + } + + #[test] + fn the_stats_report_survives_every_log_level() { + // `--stats` is an explicit request, so no `-l` may suppress it — + // including the quietest, which is the regression this guards. + for max in [ + Level::ERROR, + Level::WARN, + Level::INFO, + Level::DEBUG, + Level::TRACE, + ] { + assert!( + event_enabled(STATS_TARGET, &Level::INFO, max), + "the stats report was filtered out at -l {max}" + ); + } + } + + #[test] + fn every_other_target_still_obeys_the_level() { + // The exemption is for one target, not a hole in the filter. + assert!(!event_enabled("crust_render", &Level::INFO, Level::ERROR)); + assert!(!event_enabled( + "crust_core::scene::usd_import", + &Level::DEBUG, + Level::INFO + )); + assert!(event_enabled("crust_render", &Level::ERROR, Level::ERROR)); + assert!(event_enabled( + "crust_core::tracer", + &Level::DEBUG, + Level::DEBUG + )); + assert!(event_enabled("crust_assets", &Level::WARN, Level::INFO)); + // A near-miss on the target name is not the stats target. + assert!(!event_enabled("stats", &Level::INFO, Level::ERROR)); + // Cranelift's INFO IR dumps are filtered as DEBUG, and only those. + let jit = "cranelift_jit::backend"; + assert!(!event_enabled(jit, &Level::INFO, Level::INFO)); + assert!(event_enabled(jit, &Level::INFO, Level::DEBUG)); + assert!(event_enabled(jit, &Level::WARN, Level::INFO)); + assert!(event_enabled("crust_render", &Level::INFO, Level::INFO)); + assert!(!event_enabled( + "crust_render::stats_extra", + &Level::INFO, + Level::ERROR + )); + } + + #[test] + fn log_levels_map_one_to_one() { + assert_eq!(get_logger_level(LoggerLevel::Trace), Level::TRACE); + assert_eq!(get_logger_level(LoggerLevel::Debug), Level::DEBUG); + assert_eq!(get_logger_level(LoggerLevel::Info), Level::INFO); + assert_eq!(get_logger_level(LoggerLevel::Warn), Level::WARN); + assert_eq!(get_logger_level(LoggerLevel::Error), Level::ERROR); + } +} diff --git a/crates/crust-render/src/main.rs b/crates/crust-render/src/main.rs index 7128d04a..c25550ea 100644 --- a/crates/crust-render/src/main.rs +++ b/crates/crust-render/src/main.rs @@ -4,6 +4,11 @@ //! counting allocator) and `crust-jit` (calling generated code). #![forbid(unsafe_code)] +mod logging; +mod products; + +use logging::{LoggerLevel, STATS_TARGET}; + use clap::Parser; use crust_assets::FileAssets; use crust_core::Buffer; @@ -12,27 +17,14 @@ use crust_core::PixelFilter; use crust_core::Renderer; use crust_core::SamplingStrategy; use crust_core::Scene; +use crust_core::{AovRequest, RenderSettings}; use crust_core::{get_settings, simple_scene}; use exr::prelude::*; use indicatif::ProgressBar; use std::path::Path; use std::process::ExitCode; -use std::sync::Mutex; -use std::time::{Duration, Instant, SystemTime}; -use tracing::{Level, debug, error, info, warn}; -use tracing_subscriber::filter::filter_fn; -use tracing_subscriber::fmt; -use tracing_subscriber::layer::SubscriberExt; -use tracing_subscriber::util::SubscriberInitExt; - -#[derive(clap::ValueEnum, Clone, Debug, Copy)] -enum LoggerLevel { - Debug, - Info, - Warn, - Error, - Trace, -} +use std::time::{Duration, Instant}; +use tracing::{debug, error, info, warn}; #[derive(Parser)] #[command(version, about, long_about = None)] @@ -229,52 +221,6 @@ where .try_map(|s| s.parse::()) } -/// Target the `--stats` report is emitted under. -/// -/// It exists so the report can be exempted from `-l`: `--stats` is an -/// explicit request for the report, and honouring it only at `-l info` or -/// below would mean `--stats -l warn` silently produced nothing. The filter -/// in `main` admits this target at any level and applies `-l` to everything -/// else, which is what keeps the report a log event — reaching `--log-file` -/// like any other — without letting the log level decide whether it appears. -const STATS_TARGET: &str = "crust_render::stats"; - -/// Whether an event at `level` on `target` survives a `-l max` filter. -/// -/// Named rather than inlined into the closure so it can be tested: the whole -/// point of it is the one case that is easy to regress into silence — -/// [`STATS_TARGET`] passing at a level that rejects everything else. -fn event_enabled(target: &str, level: &Level, max: Level) -> bool { - target == STATS_TARGET || effective_level(target, level) <= max -} - -/// The level an event is filtered at, which for a few dependencies is not -/// the level it was emitted at. -/// -/// `cranelift_jit` logs the whole IR of every function it defines at INFO — -/// one multi-hundred-line dump per MaterialX program, so a default render's -/// INFO output grew with the number of materials, against the rule that INFO -/// lines do not scale with the scene. `tracing` cannot rewrite an event's -/// level, so it is *filtered* as DEBUG (shown from `-l debug` on) while still -/// printing its own `INFO` stamp. WARN and ERROR from cranelift are untouched. -fn effective_level(target: &str, level: &Level) -> Level { - if *level == Level::INFO && target.starts_with("cranelift") { - Level::DEBUG - } else { - *level - } -} - -fn get_logger_level(level: LoggerLevel) -> Level { - match level { - LoggerLevel::Debug => Level::DEBUG, - LoggerLevel::Info => Level::INFO, - LoggerLevel::Warn => Level::WARN, - LoggerLevel::Error => Level::ERROR, - LoggerLevel::Trace => Level::TRACE, - } -} - /// How the outputs describe and encode the working space's pixels: the space /// itself, for the EXRs' colour metadata, and the OCIO display / view the /// preview PNG is encoded through. @@ -403,123 +349,11 @@ fn write_beauty( Ok(()) } -/// A filename-safe UTC timestamp, `YYYYMMDDTHHMMSSZ`. -/// -/// Hand-rolled rather than pulled from `chrono` or `time`: neither is in the -/// dependency graph, and adding one to name a file would be the largest -/// dependency in this binary. `tracing-subscriber` formats its own line -/// timestamps the same way and for the same reason, so the `Z` suffix here -/// matches what the log lines themselves carry. -/// -/// The civil-from-days conversion is Howard Hinnant's, shifting the era to -/// start on 0000-03-01 so a leap day lands at the end of a 400-year cycle and -/// the month arithmetic needs no table. Valid for any date this can be handed. -fn utc_stamp(t: std::time::SystemTime) -> String { - let secs = t - .duration_since(std::time::UNIX_EPOCH) - .map(|d| d.as_secs() as i64) - // A clock before 1970 is not worth a failure path; it only names a file. - .unwrap_or(0); - let (days, rem) = (secs.div_euclid(86_400), secs.rem_euclid(86_400)); - let (hour, min, sec) = (rem / 3600, (rem % 3600) / 60, rem % 60); - - // Days since 1970-01-01 -> civil date. - let z = days + 719_468; - let era = z.div_euclid(146_097); - let doe = z.rem_euclid(146_097); - let yoe = (doe - doe / 1460 + doe / 36_524 - doe / 146_096) / 365; - let doy = doe - (365 * yoe + yoe / 4 - yoe / 100); - let mp = (5 * doy + 2) / 153; - let day = doy - (153 * mp + 2) / 5 + 1; - // `mp` counts from March; roll it back to a calendar month, and with it - // the year, which only advances once January is reached. - let month = if mp < 10 { mp + 3 } else { mp - 9 }; - let year = era * 400 + yoe + i64::from(month <= 2); - - format!("{year:04}{month:02}{day:02}T{hour:02}{min:02}{sec:02}Z") -} - -/// Opens the run's log file, creating any missing directories in `dir`. -/// -/// A failure fails the run rather than warning: nothing has been rendered yet -/// when this runs, so stopping costs no work, and a `--log-file` that quietly -/// produced no file would be discovered only after the render it was meant to -/// record. The error is the message to print. -fn open_log_file(dir: &Path) -> std::result::Result { - let path = dir.join(format!("crust-render-{}.log", utc_stamp(SystemTime::now()))); - if let Some(parent) = path.parent() - && !parent.as_os_str().is_empty() - && let Err(e) = std::fs::create_dir_all(parent) - { - return Err(format!( - "could not create log directory {}: {e}", - parent.display() - )); - } - let f = std::fs::File::create(&path) - .map_err(|e| format!("could not create log file {}: {e}", path.display()))?; - // Said on stderr rather than through `tracing`: the subscriber this file - // belongs to does not exist yet. - eprintln!("Logging to {}", path.display()); - Ok(f) -} - -/// Every failure returns through here rather than `std::process::exit`, so -/// the stack unwinds normally and every destructor runs on the way out. -fn main() -> ExitCode { - // CLI - let cli = Cli::parse(); - // Add tracing. Two layers rather than one writer teed into both, because - // ANSI is a per-layer setting: a single writer would either colour the - // file with escape codes or strip the colour from the terminal. The - // registry that composes them costs no new dependency — `sharded-slab` - // and `thread_local` are already in the graph via the `fmt` feature. - // - // The file is written unbuffered, deliberately: the subscriber that owns - // it is the process-global one, which is never dropped, so a `BufWriter` - // would never be flushed and would lose exactly the last lines — the ones - // explaining why a run stopped. A log at these volumes is not worth a - // flush guard. - let log_file = match cli.log_file.as_deref().map(open_log_file).transpose() { - Ok(f) => f, - Err(e) => { - eprintln!("error: {e}"); - return ExitCode::FAILURE; - } - }; - let level = get_logger_level(cli.level); - tracing_subscriber::registry() - // `-l` for everything except the `--stats` report, which the user - // asked for by flag and which therefore is not the log level's to - // suppress. See `STATS_TARGET`. - .with(filter_fn(move |meta| { - event_enabled(meta.target(), meta.level(), level) - })) - .with(fmt::layer()) - .with(log_file.map(|f| fmt::layer().with_ansi(false).with_writer(Mutex::new(f)))) - .init(); - let input = cli.input; - let output = cli.output; - // Built before the scene and kept until after the render: it owns the - // streaming tile cache, whose counters the `--stats` report reads once the - // last ray has been traced. - // `--ocio-config`, else `$OCIO` as every OCIO application reads it, else - // the builtin config. - let ocio = match (&cli.ocio_config, &crust_core::config().ocio) { - (Some(flag), _) => Some((flag, "--ocio-config")), - (None, Some(env)) => Some((env, "$OCIO")), - (None, None) => None, - }; - if let Some((config, from)) = ocio { - debug!("OCIO config {config} (from {from})"); - if let Err(e) = crust_core::color::use_config(config) { - error!("{from}: {e}"); - return ExitCode::FAILURE; - } - } - let assets = FileAssets::new().with_auto_tx(cli.auto_tx); - let load_start = Instant::now(); - let scene: Scene = if let Some(t) = input { +/// The scene to render: the USD stage `-i` names, imported under the CLI's +/// options, or the procedural fallback without one. A failure is already +/// logged; the error is the exit code. +fn load_scene(cli: &Cli, assets: &FileAssets) -> std::result::Result { + let scene = if let Some(t) = &cli.input { let input_path = std::path::Path::new(&t); debug!("Loading USD scene from {}", input_path.display()); let options = crust_core::UsdImportOptions { @@ -532,11 +366,11 @@ fn main() -> ExitCode { skip_stage_teardown: true, working_space: cli.working_space.clone(), }; - match Scene::from_usd_with_options(input_path, &assets, &options) { + match Scene::from_usd_with_options(input_path, assets, &options) { Ok(scene) => scene, Err(e) => { error!("Failed to load USD scene: {}", e); - return ExitCode::FAILURE; + return Err(ExitCode::FAILURE); } } } else { @@ -559,32 +393,54 @@ fn main() -> ExitCode { let (camera, settings) = get_settings(); Scene::new(camera, world, lights, settings) }; - debug!("Scene built in {:?}", load_start.elapsed()); - let output_color = match OutputColor::new(scene.working_space, &cli.display, &cli.view) { - Ok(c) => c, - Err(e) => { - error!("{e}"); - return ExitCode::FAILURE; + Ok(scene) +} + +/// The scene's render settings with the CLI's overrides applied. +fn apply_overrides(cli: &Cli, settings: RenderSettings) -> RenderSettings { + let mut settings = match cli.samples { + Some(spp) => { + debug!("--samples {spp} overrides the scene's crust:samplesPerPixel"); + settings.with_samples_per_pixel(spp) } + None => settings, }; - // One line however many textures were converted — the per-file lines are - // DEBUG, since their count grows with the stage. - let (converted, failed, secs) = assets.tx_report(); - if converted + failed > 0 { - info!("--auto-tx: converted {converted} texture tile(s) to .tx in {secs:.1}s"); - if failed > 0 { - warn!("--auto-tx: {failed} tile(s) failed to convert; their textures were preloaded"); - } + if let Some(strategy) = cli.strategy { + debug!("--strategy {strategy} overrides the scene's crust:samplingStrategy"); + settings = settings.with_sampling_strategy(strategy); } - // What to write: the stage's RenderProducts, with `-o` replacing the - // first one's path; with none, the single beauty EXR at `-o`. - let mut aovs = scene.aovs; - if let (Some(first), Some(o)) = (aovs.products.first_mut(), &output) { + if let Some(selection) = cli.light_selection { + debug!("--light-selection {selection} overrides the scene's crust:lightSelection"); + settings = settings.with_light_selection(selection); + } + // --filter replaces the scene's filter (at the filter's default radius); + // --filter-radius then resizes whichever filter is in effect, so it also + // works alone to widen the scene-authored one. + if let Some(filter) = cli.filter { + debug!("--filter {filter} overrides the scene's crust:pixelFilter"); + settings = settings.with_pixel_filter(filter); + } + if let Some(radius) = cli.filter_radius { + debug!("--filter-radius {radius} overrides the filter's own radius"); + settings = settings.with_pixel_filter(settings.pixel_filter().with_radius(radius)); + } + if let Some(limit) = cli.indirect_clamp { + debug!("--indirect-clamp {limit} overrides the scene's crust:indirectClamp"); + settings = settings.with_indirect_clamp(limit); + } + settings +} + +/// The stage's RenderProducts that can be written, with `-o` replacing the +/// first one's path; each refusal is warned about. Empty means the single +/// beauty EXR at `-o`. +fn select_products(aovs: &mut AovRequest, output: Option<&str>) { + if let (Some(first), Some(o)) = (aovs.products.first_mut(), output) { debug!( "-o {o} replaces {}'s productName {:?}", first.prim_path, first.name ); - first.name = o.clone(); + first.name = o.to_owned(); } if !aovs.products.is_empty() { aovs.products.retain(|p| { @@ -608,6 +464,62 @@ fn main() -> ExitCode { warn!("No RenderProduct can be written; writing the beauty to -o instead"); } } +} + +/// Every failure returns through here rather than `std::process::exit`, so +/// the stack unwinds normally and every destructor runs on the way out. +fn main() -> ExitCode { + // CLI + let cli = Cli::parse(); + if let Err(e) = logging::init(cli.level, cli.log_file.as_deref()) { + eprintln!("error: {e}"); + return ExitCode::FAILURE; + } + let output = cli.output.clone(); + // Built before the scene and kept until after the render: it owns the + // streaming tile cache, whose counters the `--stats` report reads once the + // last ray has been traced. + // `--ocio-config`, else `$OCIO` as every OCIO application reads it, else + // the builtin config. + let ocio = match (&cli.ocio_config, &crust_core::config().ocio) { + (Some(flag), _) => Some((flag, "--ocio-config")), + (None, Some(env)) => Some((env, "$OCIO")), + (None, None) => None, + }; + if let Some((config, from)) = ocio { + debug!("OCIO config {config} (from {from})"); + if let Err(e) = crust_core::color::use_config(config) { + error!("{from}: {e}"); + return ExitCode::FAILURE; + } + } + let assets = FileAssets::new().with_auto_tx(cli.auto_tx); + let load_start = Instant::now(); + let scene = match load_scene(&cli, &assets) { + Ok(scene) => scene, + Err(code) => return code, + }; + debug!("Scene built in {:?}", load_start.elapsed()); + let output_color = match OutputColor::new(scene.working_space, &cli.display, &cli.view) { + Ok(c) => c, + Err(e) => { + error!("{e}"); + return ExitCode::FAILURE; + } + }; + // One line however many textures were converted — the per-file lines are + // DEBUG, since their count grows with the stage. + let (converted, failed, secs) = assets.tx_report(); + if converted + failed > 0 { + info!("--auto-tx: converted {converted} texture tile(s) to .tx in {secs:.1}s"); + if failed > 0 { + warn!("--auto-tx: {failed} tile(s) failed to convert; their textures were preloaded"); + } + } + // What to write: the stage's RenderProducts, with `-o` replacing the + // first one's path; with none, the single beauty EXR at `-o`. + let mut aovs = scene.aovs; + select_products(&mut aovs, output.as_deref()); let camera = scene.camera; let world = scene.world; let lights = scene.lights; @@ -615,36 +527,7 @@ fn main() -> ExitCode { // Import phases and scene counts come from the loader; render and // output are timed here. let mut stats = scene.stats; - let mut settings = match cli.samples { - Some(spp) => { - debug!("--samples {spp} overrides the scene's crust:samplesPerPixel"); - scene.settings.with_samples_per_pixel(spp) - } - None => scene.settings, - }; - if let Some(strategy) = cli.strategy { - debug!("--strategy {strategy} overrides the scene's crust:samplingStrategy"); - settings = settings.with_sampling_strategy(strategy); - } - if let Some(selection) = cli.light_selection { - debug!("--light-selection {selection} overrides the scene's crust:lightSelection"); - settings = settings.with_light_selection(selection); - } - // --filter replaces the scene's filter (at the filter's default radius); - // --filter-radius then resizes whichever filter is in effect, so it also - // works alone to widen the scene-authored one. - if let Some(filter) = cli.filter { - debug!("--filter {filter} overrides the scene's crust:pixelFilter"); - settings = settings.with_pixel_filter(filter); - } - if let Some(radius) = cli.filter_radius { - debug!("--filter-radius {radius} overrides the filter's own radius"); - settings = settings.with_pixel_filter(settings.pixel_filter().with_radius(radius)); - } - if let Some(limit) = cli.indirect_clamp { - debug!("--indirect-clamp {limit} overrides the scene's crust:indirectClamp"); - settings = settings.with_indirect_clamp(limit); - } + let settings = apply_overrides(&cli, scene.settings); // A BVH can only cull primitives whose bounds are small against the // whole scene. Report the ratio so a scene whose instance boxes all // span everything -- where no split can help -- is visible. @@ -778,95 +661,7 @@ fn main() -> ExitCode { // only exist in a feature-on build. #[cfg(feature = "traversal-stats")] if cli.stats || cli.profile { - // Accumulated into one string and emitted as a single event, for the - // reason the report below is: a `println!` per row would leave these - // lines out of `--log-file`, and one event per row would stamp each - // of them with a timestamp the table has no column for. - use crust_core::rt::traversal_stats as ts; - use std::fmt::Write as _; - let rays = ray_stats.camera_rays.max(1) as f64; - let per = |n: u64| n as f64 / rays; - let rule = "-".repeat(84); - let mut out = String::new(); - // Infallible: `write!` into a String only fails if the formatter - // does, and none of these arguments can. - let _ = write!(out, "\n{rule}\nBVH Traversal (per camera ray)\n{rule}"); - for (level, name) in [(0usize, "top-level"), (1, "instanced")] { - let (q, nodes, leaves, packets, scalars) = ts::read_level(level); - if q == 0 { - continue; - } - let _ = write!( - out, - "\n {name:<12} queries {:>8.2} nodes {:>9.2} leaves {:>8.2} packets {:>7.2} scalar {:>8.2}", - per(q), - per(nodes), - per(leaves), - per(packets), - per(scalars), - ); - } - // Which top-level instances the descents went into. A top level that - // culls well spreads them thinly; one that does not concentrates them - // on whatever geometry every ray's path overlaps. The importer's - // DEBUG lines give each instancer's `geom ids a..b` range, which is - // how an id here is traced back to a prim. - let descents = ts::top_level_descents(); - let total: u64 = descents.iter().map(|d| d.1).sum(); - if total > 0 { - let mut acc = 0u64; - let mut marks = vec![]; - for (i, d) in descents.iter().enumerate() { - acc += d.1; - for f in [0.5, 0.9, 0.99] { - if (acc as f64) >= f * total as f64 && !marks.iter().any(|&(g, _)| g == f) { - marks.push((f, i + 1)); - } - } - } - let _ = write!( - out, - "\n top-level instances entered: {} of them, {:.1} descents per camera ray \ - (closest-hit and shadow rays; the rows above count closest-hit only)", - descents.len(), - per(total) - ); - for (f, n) in marks { - let _ = write!( - out, - "\n {:.0}% of descents go to {n} instances", - f * 100.0 - ); - } - let top: Vec<_> = descents.iter().take(40).collect(); - let ids: std::collections::HashSet = top.iter().map(|d| d.0).collect(); - let info: std::collections::HashMap = renderer - .world - .describe_instances(&ids) - .into_iter() - .map(|(id, b, n, shared)| (id, (b, n, shared))) - .collect(); - let _ = write!( - out, - "\n {:>9} {:>7} {:>9} {:>8} {:>9} bounds", - "geom_id", "share", "per ray", "prims", "shared by" - ); - for &&(id, n) in &top { - let (b, prims, shared) = info[&id]; - let _ = write!( - out, - "\n {id:>9} {:>6.2}% {:>9.2} {prims:>8} {shared:>9} [{:.0} {:.0} {:.0}]..[{:.0} {:.0} {:.0}]", - 100.0 * n as f64 / total as f64, - per(n), - b.minimum.x, - b.minimum.y, - b.minimum.z, - b.maximum.x, - b.maximum.y, - b.maximum.z, - ); - } - } + let out = crust_core::traversal_report(&renderer.world, ray_stats.camera_rays); info!(target: STATS_TARGET, "{out}"); } @@ -890,54 +685,6 @@ fn main() -> ExitCode { mod tests { use super::*; - /// `SystemTime` at a given Unix second, for pinning `utc_stamp` against - /// dates whose answers are known independently. - fn at(unix_secs: u64) -> SystemTime { - std::time::UNIX_EPOCH + std::time::Duration::from_secs(unix_secs) - } - - #[test] - fn utc_stamp_names_known_instants() { - assert_eq!(utc_stamp(at(0)), "19700101T000000Z"); - assert_eq!(utc_stamp(at(1_774_267_884)), "20260323T121124Z"); - // Last second of a year, and the first of the next. - assert_eq!(utc_stamp(at(1_767_225_599)), "20251231T235959Z"); - assert_eq!(utc_stamp(at(1_767_225_600)), "20260101T000000Z"); - } - - #[test] - fn utc_stamp_handles_leap_years() { - // 2024 is a leap year: Feb 29 exists. - assert_eq!(utc_stamp(at(1_709_164_800)), "20240229T000000Z"); - // 2000 is a leap year (divisible by 400) — the case a naive - // "divisible by 4, except by 100" rule gets wrong. - assert_eq!(utc_stamp(at(951_782_400)), "20000229T000000Z"); - // 1900 was NOT a leap year, but it predates the epoch, so check the - // other end of the same rule: 2100 is not one either, and March 1 - // must follow February 28. - assert_eq!(utc_stamp(at(4_107_456_000)), "21000228T000000Z"); - assert_eq!(utc_stamp(at(4_107_542_400)), "21000301T000000Z"); - } - - #[test] - fn utc_stamp_is_filename_safe_and_sorts_chronologically() { - let mut prev = utc_stamp(at(0)); - for day in 1..4000u64 { - // Every 37 days, so the walk crosses month and year boundaries - // at varied offsets rather than landing on the same day each time. - let t = utc_stamp(at(day * 37 * 86_400 + 3661)); - assert!( - t.chars().all(|c| c.is_ascii_alphanumeric()), - "{t} is not filename-safe" - ); - assert_eq!(t.len(), 16, "{t} is not a fixed-width stamp"); - // Fixed width and zero-padded, so lexical order is chronological - // — which is the whole reason for this format over a locale one. - assert!(t > prev, "{t} does not sort after {prev}"); - prev = t; - } - } - #[test] fn log_file_flag_is_optional_and_takes_an_optional_directory() { // Absent: no file. @@ -959,55 +706,6 @@ mod tests { assert!(c.bucket); } - #[test] - fn the_stats_report_survives_every_log_level() { - // `--stats` is an explicit request, so no `-l` may suppress it — - // including the quietest, which is the regression this guards. - for max in [ - Level::ERROR, - Level::WARN, - Level::INFO, - Level::DEBUG, - Level::TRACE, - ] { - assert!( - event_enabled(STATS_TARGET, &Level::INFO, max), - "the stats report was filtered out at -l {max}" - ); - } - } - - #[test] - fn every_other_target_still_obeys_the_level() { - // The exemption is for one target, not a hole in the filter. - assert!(!event_enabled("crust_render", &Level::INFO, Level::ERROR)); - assert!(!event_enabled( - "crust_core::scene::usd_import", - &Level::DEBUG, - Level::INFO - )); - assert!(event_enabled("crust_render", &Level::ERROR, Level::ERROR)); - assert!(event_enabled( - "crust_core::tracer", - &Level::DEBUG, - Level::DEBUG - )); - assert!(event_enabled("crust_assets", &Level::WARN, Level::INFO)); - // A near-miss on the target name is not the stats target. - assert!(!event_enabled("stats", &Level::INFO, Level::ERROR)); - // Cranelift's INFO IR dumps are filtered as DEBUG, and only those. - let jit = "cranelift_jit::backend"; - assert!(!event_enabled(jit, &Level::INFO, Level::INFO)); - assert!(event_enabled(jit, &Level::INFO, Level::DEBUG)); - assert!(event_enabled(jit, &Level::WARN, Level::INFO)); - assert!(event_enabled("crust_render", &Level::INFO, Level::INFO)); - assert!(!event_enabled( - "crust_render::stats_extra", - &Level::INFO, - Level::ERROR - )); - } - /// The default outputs: `lin_rec709`, previewed un-tone-mapped on sRGB. pub(crate) fn rec709() -> OutputColor { let w = crust_core::color::Space::LIN_REC709; @@ -1146,15 +844,6 @@ mod tests { ); } - #[test] - fn log_levels_map_one_to_one() { - assert_eq!(get_logger_level(LoggerLevel::Trace), Level::TRACE); - assert_eq!(get_logger_level(LoggerLevel::Debug), Level::DEBUG); - assert_eq!(get_logger_level(LoggerLevel::Info), Level::INFO); - assert_eq!(get_logger_level(LoggerLevel::Warn), Level::WARN); - assert_eq!(get_logger_level(LoggerLevel::Error), Level::ERROR); - } - #[test] fn write_png_flips_rows_and_tone_maps() { let (w, h) = (3usize, 2usize); @@ -1326,387 +1015,3 @@ mod tests { assert!(Cli::try_parse_from(["crust-render", "-s", "many"]).is_err()); } } - -// Inline rather than a file of its own: the CLI crate keeps one source file. -mod products { - //! Writing a render's `RenderProduct`s: one single-part, scanline, - //! ZIP16-compressed EXR per product, one layer of named channels per var. - //! - //! Channel names follow OpenEXR's `.` convention and the - //! ASWF Color Interop rule that only colour gets `R/G/B`: colour vars are - //! `.R/.G/.B[/.A]`, vectors `.X/.Y/.Z`, UVs `.U/.V`, and - //! a scalar is one channel named after its layer. The product's first beauty - //! var is written bare (`R/G/B[/A]`) so every viewer shows it as the image. - //! - //! Scanline, not the `exr` crate's default tiling: tinyexr crashes on crust's - //! tiled files (`docs/material_fidelity.md`). The no-products render does not - //! come through here at all — it keeps `write_rgb_file`, byte for byte. - - use crust_core::{AovFilm, AovProduct, AovVar, Buffer, ChannelKind, Precision}; - use exr::meta::attribute::AttributeValue; - use exr::prelude::{ - AnyChannel, AnyChannels, Blocks, Compression, Encoding, FlatSamples, Image, Layer, - LayerAttributes, LineOrder, Text, WritableImage, f16, - }; - use std::io; - use std::path::Path; - use tracing::warn; - - /// The Color Interop ID for a working space without one in the config. - /// The file still says what it can: its chromaticities, when known. - const UNKNOWN_INTEROP_ID: &str = "unknown"; - - /// The channel names `var` writes, one per plane `AovFilm::var_channels` - /// returns. `bare` makes the layer prefix empty — the product's beauty. - pub fn channel_names(var: &AovVar, bare: bool) -> Vec { - let layer = match &var.channel_prefix { - Some(prefix) => prefix.clone(), - None if bare => String::new(), - None => var.name.clone(), - }; - let components: &[&str] = match var.source.channel_kind() { - ChannelKind::Color if var.with_alpha() => &["R", "G", "B", "A"], - ChannelKind::Color => &["R", "G", "B"], - ChannelKind::Vector => &["X", "Y", "Z"], - ChannelKind::Uv => &["U", "V"], - ChannelKind::Scalar => { - return vec![if layer.is_empty() { - var.name.clone() - } else { - layer - }]; - } - }; - components - .iter() - .map(|c| { - if layer.is_empty() { - (*c).to_owned() - } else { - format!("{layer}.{c}") - } - }) - .collect() - } - - /// `plane` as `precision` samples. A UINT channel holds integers; a negative - /// one (the `-1` "no ID" clear value) is written as its two's-complement - /// bit pattern. - fn samples(plane: Vec, precision: Precision) -> FlatSamples { - match precision { - Precision::Half => FlatSamples::F16(plane.into_iter().map(f16::from_f32).collect()), - Precision::Float => FlatSamples::F32(plane), - Precision::Uint => { - FlatSamples::U32(plane.into_iter().map(|v| v as i32 as u32).collect()) - } - } - } - - /// The layout of a product's file: its channels' names, in var order, with - /// the var each comes from. A var whose names collide with an earlier one's - /// is refused (warned) rather than written over it. - pub fn product_channels(product: &AovProduct) -> Vec<(&AovVar, Vec)> { - let beauty = product.beauty().map(|v| v.prim_path.as_str()); - let mut taken: Vec = Vec::new(); - let mut out = Vec::new(); - for var in &product.vars { - let bare = Some(var.prim_path.as_str()) == beauty; - let names = channel_names(var, bare); - if let Some(clash) = names.iter().find(|n| taken.contains(n)) { - warn!( - "{}: channel {clash:?} is already written by another var of {}; skipped", - var.prim_path, product.prim_path - ); - continue; - } - taken.extend(names.iter().cloned()); - out.push((var, names)); - } - out - } - - /// Refuses (with a warning) every product whose path is one an earlier - /// product already writes, keeping the first: the later write would replace - /// the earlier file and silently lose its channels. Paths are compared - /// lexically, component by component (`a/./b.exr` is `a/b.exr`), since the - /// files need not exist yet. - pub fn refuse_shared_paths(products: &mut Vec) { - let mut seen: Vec<(std::path::PathBuf, String)> = Vec::new(); - products.retain(|p| { - let path: std::path::PathBuf = Path::new(&p.name).components().collect(); - if let Some((_, first)) = seen.iter().find(|(q, _)| *q == path) { - warn!( - "{} writes {}, as {first} already does; nothing written for it", - p.prim_path, - path.display() - ); - return false; - } - seen.push((path, p.prim_path.clone())); - true - }); - } - - /// Writes `product` to `path`, creating its parent directories. Returns the - /// channel names written, in file (alphabetical) order. - /// - /// The colour channels are tagged with the working space's ASWF Color - /// Interop ID (`colorInteropID`, `docs/color_management.md`) and, off - /// Rec.709, its chromaticities. - pub fn write_product( - path: &Path, - product: &AovProduct, - beauty: &Buffer, - film: &AovFilm, - color: &super::OutputColor, - ) -> io::Result> { - let interop = crust_core::color::interop_id(color.working); - let interop = interop.as_deref().unwrap_or(UNKNOWN_INTEROP_ID); - let (width, height) = film.dimensions(); - let mut channels = Vec::new(); - for (var, names) in product_channels(product) { - for (name, plane) in names.iter().zip(film.var_channels(beauty, var)) { - let name = Text::new_or_none(name).ok_or_else(|| { - io::Error::new( - io::ErrorKind::InvalidInput, - format!("channel name {name:?} is not valid in an EXR"), - ) - })?; - channels.push(AnyChannel::new(name, samples(plane, var.precision))); - } - } - if channels.is_empty() { - return Err(io::Error::new( - io::ErrorKind::InvalidInput, - "no channel to write", - )); - } - if let Some(dir) = path.parent().filter(|d| !d.as_os_str().is_empty()) { - std::fs::create_dir_all(dir)?; - } - // Sorted, because EXR stores channels alphabetically and a reader finds - // them by name — an unsorted list is a malformed file. - let channels = AnyChannels::sort(channels.into_iter().collect()); - let names = channels.list.iter().map(|c| c.name.to_string()).collect(); - - let mut attributes = LayerAttributes { - software_name: Text::new_or_none(concat!("crust-render ", env!("CARGO_PKG_VERSION"))), - ..LayerAttributes::default() - }; - let mut text = vec![("colorInteropID", interop)]; - for (key, value) in &product.attributes { - match key.as_str() { - // Standard attributes `exr` exposes as typed fields. - "comments" | "comment" => attributes.comments = Text::new_or_none(value), - "owner" => attributes.owner = Text::new_or_none(value), - // Describes the pixels crust wrote, so only crust may set it. - "colorInteropID" => warn!( - "{}: driver:parameters colorInteropID = {value:?} is not copied: the colour \ - channels are {interop}", - product.prim_path - ), - k if exr::meta::header::standard_names::ALL.contains(&k.as_bytes()) => { - warn!( - "{}: driver:parameters {k:?} names a standard EXR attribute crust sets \ - itself; not copied", - product.prim_path - ); - } - k => text.push((k, value)), - } - } - for (key, value) in text { - if let (Some(k), Some(v)) = (Text::new_or_none(key), Text::new_or_none(value)) { - attributes.other.insert(k, AttributeValue::Text(v)); - } - } - - let layer = Layer::new( - (width, height), - attributes, - Encoding { - compression: Compression::ZIP16, - blocks: Blocks::ScanLines, - line_order: LineOrder::Increasing, - }, - channels, - ); - let mut image = Image::from_layer(layer); - image.attributes.chromaticities = color.exr_chromaticities(); - image - .write() - .to_file(path) - .map_err(|e| io::Error::other(format!("{}: {e}", path.display())))?; - Ok(names) - } - - #[cfg(test)] - mod tests { - use super::*; - use crust_core::{Accumulation, AovSource}; - use exr::prelude::{ReadChannels, ReadLayers, read}; - - fn var(name: &str, source: AovSource) -> AovVar { - AovVar { - prim_path: format!("/Render/Vars/{name}"), - name: name.to_owned(), - channel_prefix: None, - source, - components: source.components(), - precision: Precision::Float, - accumulation: Accumulation::Filtered, - clear: source.default_clear(), - expression: None, - raw: false, - } - } - - fn product(vars: Vec) -> AovProduct { - AovProduct { - prim_path: "/Render/p".into(), - name: "p.exr".into(), - vars, - attributes: vec![ - ("artist".into(), "someone".into()), - ("colorInteropID".into(), "srgb_rec709_display".into()), - ("comments".into(), "a note".into()), - ], - } - } - - fn names(p: &AovProduct) -> Vec { - product_channels(p) - .into_iter() - .flat_map(|(_, names)| names) - .collect() - } - - #[test] - fn channels_follow_the_layer_dot_component_convention() { - let mut beauty = var("beauty", AovSource::Color); - beauty.components = 4; - let p = product(vec![ - beauty, - var("Z", AovSource::Depth), - var("N", AovSource::Normal), - var("st", AovSource::St), - var("diffuse", AovSource::Color), - ]); - // The beauty bare, scalars named after their layer, data in X/Y/Z or - // U/V, a second colour var prefixed. - assert_eq!( - names(&p), - [ - "R", - "G", - "B", - "A", - "Z", - "N.X", - "N.Y", - "N.Z", - "st.U", - "st.V", - "diffuse.R", - "diffuse.G", - "diffuse.B" - ] - ); - } - - #[test] - fn a_channel_prefix_replaces_the_layer_and_a_clash_is_refused() { - let mut beauty = var("beauty", AovSource::Color); - beauty.channel_prefix = Some("rgba".into()); - let mut p_world = var("P", AovSource::P); - p_world.channel_prefix = Some("Pw".into()); - let clash = var("Z", AovSource::Depth); - let p = product(vec![beauty, p_world, var("Z", AovSource::Depth), clash]); - assert_eq!( - names(&p), - ["rgba.R", "rgba.G", "rgba.B", "Pw.X", "Pw.Y", "Pw.Z", "Z"] - ); - } - - #[test] - fn a_product_is_written_scanline_zip_with_its_header() { - let (w, h) = (3, 2); - let mut beauty = Buffer::new(w, h); - beauty.set_pixel(0, h - 1, crust_core::Vec3A::new(1.0, 2.0, 3.0)); - let film = crust_core::AovFilm::empty(w, h); - let dir = std::env::temp_dir().join("crust_render_products_test/nested"); - let _ = std::fs::remove_dir_all(&dir); - let path = dir.join("p.exr"); - // A beauty-only product comes out of an empty film; the parent - // directories do not exist yet. - let p = product(vec![var("beauty", AovSource::Color)]); - let rec709 = crate::tests::rec709(); - let written = write_product(&path, &p, &beauty, &film, &rec709).expect("written"); - assert_eq!(written, ["B", "G", "R"]); - let image = read() - .no_deep_data() - .largest_resolution_level() - .all_channels() - .first_valid_layer() - .all_attributes() - .from_file(&path) - .expect("reads back"); - let layer = &image.layer_data; - assert_eq!(layer.encoding.blocks, Blocks::ScanLines); - assert_eq!(layer.encoding.compression, Compression::ZIP16); - assert_eq!( - layer.attributes.other.get(&Text::from("colorInteropID")), - Some(&AttributeValue::Text(Text::from("lin_rec709_scene"))) - ); - assert_eq!( - layer.attributes.other.get(&Text::from("artist")), - Some(&AttributeValue::Text(Text::from("someone"))) - ); - assert_eq!(layer.attributes.comments, Some(Text::from("a note"))); - // An authored colorInteropID does not replace crust's own: the - // pixels are linear Rec.709 whatever the product says. - assert_eq!( - layer.attributes.other.get(&Text::from("colorInteropID")), - Some(&AttributeValue::Text(Text::from("lin_rec709_scene"))) - ); - // Top-down rows: the buffer's top row (y = h - 1) is the file's first. - let r = &layer.channel_data.list[2]; - assert_eq!(r.name, Text::from("R")); - assert_eq!(r.sample_data.value_by_flat_index(0).to_f32(), 1.0); - } - - #[test] - fn a_product_sharing_an_earlier_path_is_refused() { - let named = |prim: &str, name: &str| AovProduct { - prim_path: prim.into(), - name: name.into(), - ..product(vec![var("beauty", AovSource::Color)]) - }; - let mut products = vec![ - named("/a", "out/a.exr"), - named("/b", "out/./a.exr"), - named("/c", "out/c.exr"), - ]; - refuse_shared_paths(&mut products); - let kept: Vec<_> = products.iter().map(|p| p.prim_path.as_str()).collect(); - assert_eq!(kept, ["/a", "/c"]); - } - - #[test] - fn samples_take_the_requested_precision() { - assert!(matches!( - samples(vec![0.5], Precision::Half), - FlatSamples::F16(v) if v == [f16::from_f32(0.5)] - )); - assert!(matches!( - samples(vec![0.5], Precision::Float), - FlatSamples::F32(v) if v == [0.5] - )); - // `-1` ("no ID") is the all-ones bit pattern. - assert!(matches!( - samples(vec![-1.0, 7.0], Precision::Uint), - FlatSamples::U32(v) if v == [0xFFFF_FFFF, 7] - )); - } - } -} diff --git a/crates/crust-render/src/products.rs b/crates/crust-render/src/products.rs new file mode 100644 index 00000000..90c9c971 --- /dev/null +++ b/crates/crust-render/src/products.rs @@ -0,0 +1,378 @@ +//! Writing a render's `RenderProduct`s: one single-part, scanline, +//! ZIP16-compressed EXR per product, one layer of named channels per var. +//! +//! Channel names follow OpenEXR's `.` convention and the +//! ASWF Color Interop rule that only colour gets `R/G/B`: colour vars are +//! `.R/.G/.B[/.A]`, vectors `.X/.Y/.Z`, UVs `.U/.V`, and +//! a scalar is one channel named after its layer. The product's first beauty +//! var is written bare (`R/G/B[/A]`) so every viewer shows it as the image. +//! +//! Scanline, not the `exr` crate's default tiling: tinyexr crashes on crust's +//! tiled files (`docs/material_fidelity.md`). The no-products render does not +//! come through here at all — it keeps `write_rgb_file`, byte for byte. + +use crust_core::{AovFilm, AovProduct, AovVar, Buffer, ChannelKind, Precision}; +use exr::meta::attribute::AttributeValue; +use exr::prelude::{ + AnyChannel, AnyChannels, Blocks, Compression, Encoding, FlatSamples, Image, Layer, + LayerAttributes, LineOrder, Text, WritableImage, f16, +}; +use std::io; +use std::path::Path; +use tracing::warn; + +/// The Color Interop ID for a working space without one in the config. +/// The file still says what it can: its chromaticities, when known. +const UNKNOWN_INTEROP_ID: &str = "unknown"; + +/// The channel names `var` writes, one per plane `AovFilm::var_channels` +/// returns. `bare` makes the layer prefix empty — the product's beauty. +pub fn channel_names(var: &AovVar, bare: bool) -> Vec { + let layer = match &var.channel_prefix { + Some(prefix) => prefix.clone(), + None if bare => String::new(), + None => var.name.clone(), + }; + let components: &[&str] = match var.source.channel_kind() { + ChannelKind::Color if var.with_alpha() => &["R", "G", "B", "A"], + ChannelKind::Color => &["R", "G", "B"], + ChannelKind::Vector => &["X", "Y", "Z"], + ChannelKind::Uv => &["U", "V"], + ChannelKind::Scalar => { + return vec![if layer.is_empty() { + var.name.clone() + } else { + layer + }]; + } + }; + components + .iter() + .map(|c| { + if layer.is_empty() { + (*c).to_owned() + } else { + format!("{layer}.{c}") + } + }) + .collect() +} + +/// `plane` as `precision` samples. A UINT channel holds integers; a negative +/// one (the `-1` "no ID" clear value) is written as its two's-complement +/// bit pattern. +fn samples(plane: Vec, precision: Precision) -> FlatSamples { + match precision { + Precision::Half => FlatSamples::F16(plane.into_iter().map(f16::from_f32).collect()), + Precision::Float => FlatSamples::F32(plane), + Precision::Uint => FlatSamples::U32(plane.into_iter().map(|v| v as i32 as u32).collect()), + } +} + +/// The layout of a product's file: its channels' names, in var order, with +/// the var each comes from. A var whose names collide with an earlier one's +/// is refused (warned) rather than written over it. +pub fn product_channels(product: &AovProduct) -> Vec<(&AovVar, Vec)> { + let beauty = product.beauty().map(|v| v.prim_path.as_str()); + let mut taken: Vec = Vec::new(); + let mut out = Vec::new(); + for var in &product.vars { + let bare = Some(var.prim_path.as_str()) == beauty; + let names = channel_names(var, bare); + if let Some(clash) = names.iter().find(|n| taken.contains(n)) { + warn!( + "{}: channel {clash:?} is already written by another var of {}; skipped", + var.prim_path, product.prim_path + ); + continue; + } + taken.extend(names.iter().cloned()); + out.push((var, names)); + } + out +} + +/// Refuses (with a warning) every product whose path is one an earlier +/// product already writes, keeping the first: the later write would replace +/// the earlier file and silently lose its channels. Paths are compared +/// lexically, component by component (`a/./b.exr` is `a/b.exr`), since the +/// files need not exist yet. +pub fn refuse_shared_paths(products: &mut Vec) { + let mut seen: Vec<(std::path::PathBuf, String)> = Vec::new(); + products.retain(|p| { + let path: std::path::PathBuf = Path::new(&p.name).components().collect(); + if let Some((_, first)) = seen.iter().find(|(q, _)| *q == path) { + warn!( + "{} writes {}, as {first} already does; nothing written for it", + p.prim_path, + path.display() + ); + return false; + } + seen.push((path, p.prim_path.clone())); + true + }); +} + +/// Writes `product` to `path`, creating its parent directories. Returns the +/// channel names written, in file (alphabetical) order. +/// +/// The colour channels are tagged with the working space's ASWF Color +/// Interop ID (`colorInteropID`, `docs/color_management.md`) and, off +/// Rec.709, its chromaticities. +pub fn write_product( + path: &Path, + product: &AovProduct, + beauty: &Buffer, + film: &AovFilm, + color: &super::OutputColor, +) -> io::Result> { + let interop = crust_core::color::interop_id(color.working); + let interop = interop.as_deref().unwrap_or(UNKNOWN_INTEROP_ID); + let (width, height) = film.dimensions(); + let mut channels = Vec::new(); + for (var, names) in product_channels(product) { + for (name, plane) in names.iter().zip(film.var_channels(beauty, var)) { + let name = Text::new_or_none(name).ok_or_else(|| { + io::Error::new( + io::ErrorKind::InvalidInput, + format!("channel name {name:?} is not valid in an EXR"), + ) + })?; + channels.push(AnyChannel::new(name, samples(plane, var.precision))); + } + } + if channels.is_empty() { + return Err(io::Error::new( + io::ErrorKind::InvalidInput, + "no channel to write", + )); + } + if let Some(dir) = path.parent().filter(|d| !d.as_os_str().is_empty()) { + std::fs::create_dir_all(dir)?; + } + // Sorted, because EXR stores channels alphabetically and a reader finds + // them by name — an unsorted list is a malformed file. + let channels = AnyChannels::sort(channels.into_iter().collect()); + let names = channels.list.iter().map(|c| c.name.to_string()).collect(); + + let mut attributes = LayerAttributes { + software_name: Text::new_or_none(concat!("crust-render ", env!("CARGO_PKG_VERSION"))), + ..LayerAttributes::default() + }; + let mut text = vec![("colorInteropID", interop)]; + for (key, value) in &product.attributes { + match key.as_str() { + // Standard attributes `exr` exposes as typed fields. + "comments" | "comment" => attributes.comments = Text::new_or_none(value), + "owner" => attributes.owner = Text::new_or_none(value), + // Describes the pixels crust wrote, so only crust may set it. + "colorInteropID" => warn!( + "{}: driver:parameters colorInteropID = {value:?} is not copied: the colour \ + channels are {interop}", + product.prim_path + ), + k if exr::meta::header::standard_names::ALL.contains(&k.as_bytes()) => { + warn!( + "{}: driver:parameters {k:?} names a standard EXR attribute crust sets \ + itself; not copied", + product.prim_path + ); + } + k => text.push((k, value)), + } + } + for (key, value) in text { + if let (Some(k), Some(v)) = (Text::new_or_none(key), Text::new_or_none(value)) { + attributes.other.insert(k, AttributeValue::Text(v)); + } + } + + let layer = Layer::new( + (width, height), + attributes, + Encoding { + compression: Compression::ZIP16, + blocks: Blocks::ScanLines, + line_order: LineOrder::Increasing, + }, + channels, + ); + let mut image = Image::from_layer(layer); + image.attributes.chromaticities = color.exr_chromaticities(); + image + .write() + .to_file(path) + .map_err(|e| io::Error::other(format!("{}: {e}", path.display())))?; + Ok(names) +} + +#[cfg(test)] +mod tests { + use super::*; + use crust_core::{Accumulation, AovSource}; + use exr::prelude::{ReadChannels, ReadLayers, read}; + + fn var(name: &str, source: AovSource) -> AovVar { + AovVar { + prim_path: format!("/Render/Vars/{name}"), + name: name.to_owned(), + channel_prefix: None, + source, + components: source.components(), + precision: Precision::Float, + accumulation: Accumulation::Filtered, + clear: source.default_clear(), + expression: None, + raw: false, + } + } + + fn product(vars: Vec) -> AovProduct { + AovProduct { + prim_path: "/Render/p".into(), + name: "p.exr".into(), + vars, + attributes: vec![ + ("artist".into(), "someone".into()), + ("colorInteropID".into(), "srgb_rec709_display".into()), + ("comments".into(), "a note".into()), + ], + } + } + + fn names(p: &AovProduct) -> Vec { + product_channels(p) + .into_iter() + .flat_map(|(_, names)| names) + .collect() + } + + #[test] + fn channels_follow_the_layer_dot_component_convention() { + let mut beauty = var("beauty", AovSource::Color); + beauty.components = 4; + let p = product(vec![ + beauty, + var("Z", AovSource::Depth), + var("N", AovSource::Normal), + var("st", AovSource::St), + var("diffuse", AovSource::Color), + ]); + // The beauty bare, scalars named after their layer, data in X/Y/Z or + // U/V, a second colour var prefixed. + assert_eq!( + names(&p), + [ + "R", + "G", + "B", + "A", + "Z", + "N.X", + "N.Y", + "N.Z", + "st.U", + "st.V", + "diffuse.R", + "diffuse.G", + "diffuse.B" + ] + ); + } + + #[test] + fn a_channel_prefix_replaces_the_layer_and_a_clash_is_refused() { + let mut beauty = var("beauty", AovSource::Color); + beauty.channel_prefix = Some("rgba".into()); + let mut p_world = var("P", AovSource::P); + p_world.channel_prefix = Some("Pw".into()); + let clash = var("Z", AovSource::Depth); + let p = product(vec![beauty, p_world, var("Z", AovSource::Depth), clash]); + assert_eq!( + names(&p), + ["rgba.R", "rgba.G", "rgba.B", "Pw.X", "Pw.Y", "Pw.Z", "Z"] + ); + } + + #[test] + fn a_product_is_written_scanline_zip_with_its_header() { + let (w, h) = (3, 2); + let mut beauty = Buffer::new(w, h); + beauty.set_pixel(0, h - 1, crust_core::Vec3A::new(1.0, 2.0, 3.0)); + let film = crust_core::AovFilm::empty(w, h); + let dir = std::env::temp_dir().join("crust_render_products_test/nested"); + let _ = std::fs::remove_dir_all(&dir); + let path = dir.join("p.exr"); + // A beauty-only product comes out of an empty film; the parent + // directories do not exist yet. + let p = product(vec![var("beauty", AovSource::Color)]); + let rec709 = crate::tests::rec709(); + let written = write_product(&path, &p, &beauty, &film, &rec709).expect("written"); + assert_eq!(written, ["B", "G", "R"]); + let image = read() + .no_deep_data() + .largest_resolution_level() + .all_channels() + .first_valid_layer() + .all_attributes() + .from_file(&path) + .expect("reads back"); + let layer = &image.layer_data; + assert_eq!(layer.encoding.blocks, Blocks::ScanLines); + assert_eq!(layer.encoding.compression, Compression::ZIP16); + assert_eq!( + layer.attributes.other.get(&Text::from("colorInteropID")), + Some(&AttributeValue::Text(Text::from("lin_rec709_scene"))) + ); + assert_eq!( + layer.attributes.other.get(&Text::from("artist")), + Some(&AttributeValue::Text(Text::from("someone"))) + ); + assert_eq!(layer.attributes.comments, Some(Text::from("a note"))); + // An authored colorInteropID does not replace crust's own: the + // pixels are linear Rec.709 whatever the product says. + assert_eq!( + layer.attributes.other.get(&Text::from("colorInteropID")), + Some(&AttributeValue::Text(Text::from("lin_rec709_scene"))) + ); + // Top-down rows: the buffer's top row (y = h - 1) is the file's first. + let r = &layer.channel_data.list[2]; + assert_eq!(r.name, Text::from("R")); + assert_eq!(r.sample_data.value_by_flat_index(0).to_f32(), 1.0); + } + + #[test] + fn a_product_sharing_an_earlier_path_is_refused() { + let named = |prim: &str, name: &str| AovProduct { + prim_path: prim.into(), + name: name.into(), + ..product(vec![var("beauty", AovSource::Color)]) + }; + let mut products = vec![ + named("/a", "out/a.exr"), + named("/b", "out/./a.exr"), + named("/c", "out/c.exr"), + ]; + refuse_shared_paths(&mut products); + let kept: Vec<_> = products.iter().map(|p| p.prim_path.as_str()).collect(); + assert_eq!(kept, ["/a", "/c"]); + } + + #[test] + fn samples_take_the_requested_precision() { + assert!(matches!( + samples(vec![0.5], Precision::Half), + FlatSamples::F16(v) if v == [f16::from_f32(0.5)] + )); + assert!(matches!( + samples(vec![0.5], Precision::Float), + FlatSamples::F32(v) if v == [0.5] + )); + // `-1` ("no ID") is the all-ones bit pattern. + assert!(matches!( + samples(vec![-1.0, 7.0], Precision::Uint), + FlatSamples::U32(v) if v == [0xFFFF_FFFF, 7] + )); + } +} diff --git a/crates/crust-rt/src/scene.rs b/crates/crust-rt/src/scene.rs deleted file mode 100644 index c15e7176..00000000 --- a/crates/crust-rt/src/scene.rs +++ /dev/null @@ -1,1833 +0,0 @@ -//! The Embree-shaped public API: attach [`Geometry`] objects to a -//! [`SceneBuilder`], `commit()` into an immutable [`Scene`], query with -//! `intersect` / `occluded`. - -use crate::aabb::AABB; -use crate::bvh::{Bvh, Primitives}; -use crate::prim::{ - CubicCurvePrim, CurvePrim, CylinderPrim, DEGENERATE_VERTEX, DiskPrim, GeomTable, - InstanceMotion, InstancePrim, NO_ID_OFFSET, PrimHit, PrimNode, SpherePrim, TriangleRecord, - transformed_aabb, -}; -use crate::ray::{MASK_ALL, Ray, RayMask}; -use glam::{Affine3A, Vec3A}; -use std::sync::Arc; - -/// One round (sphere-swept) curve segment: a cone frustum tangent to the -/// spheres `(p0, r0)` and `(p1, r1)`, with spherical caps. -#[derive(Clone, Copy, Debug)] -pub struct CurveSegment { - pub p0: Vec3A, - pub p1: Vec3A, - pub r0: f32, - pub r1: f32, -} - -/// How a committed scene stores its triangle packets — see -/// [`SceneBuilder::commit_with`]. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub enum PacketLayout { - /// Each packet carries its four triangles' vertices (192 bytes): the - /// fastest in cache. The layout before indexed packets existed. - Gathered, - /// Each packet carries vertex indices (92 bytes) and gathers from the - /// scene's shared vertex table at every test; bit-identical hits. A - /// memory trade, not a speed one: on the subdivision stress grid it is - /// 104 → 79 kernel bytes per triangle for 13% less render throughput, - /// and on a 4 M-triangle soup that does not fit the cache it is 22% - /// fewer kernel bytes for 30% slower traversal — the twelve dependent - /// vertex loads per packet cost more than the bandwidth they save. - Indexed, - /// The measured default: `Gathered`. The expectation was that a tree - /// too large for the cache would favour the smaller packet; measured - /// with `ray_throughput --layout` in and out of cache, no size does, so - /// the indexed layout is an explicit opt-in for a scene that otherwise - /// does not fit. - #[default] - Auto, -} - -/// What [`SceneBuilder::commit_with`] lets a caller choose about the build. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct CommitOptions { - /// The triangle packet layout. - pub layout: PacketLayout, -} - -impl Default for CommitOptions { - fn default() -> Self { - CommitOptions { - layout: PacketLayout::Auto, - } - } -} - -/// Exact bytes a committed [`Scene`] holds, by structure — the kernel's -/// side of a memory report. Counts `capacity`, not `len`, because unused -/// capacity is resident too, and deduplicates shared instanced scenes so -/// a prototype placed a thousand times is counted once. -/// -/// Only the kernel's own allocations: the application's material tables, -/// the USD stage and anything else outside `crust-rt` are not visible -/// from here. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub struct MemoryFootprint { - /// The `PrimNode` arrays: spheres, disks, cylinders and linear curve - /// segments. - pub prim_nodes: usize, - /// Instances, 96 bytes each inline, plus the endpoint transforms of the - /// moving ones. - pub instances: usize, - /// Cubic curve spans, 96 bytes each inline. - pub cubic_spans: usize, - /// 4-wide BVH nodes. - pub bvh_nodes: usize, - pub leaves: usize, - /// Triangle SIMD packets of the gathered layout (the vertices, SoA). - pub packets: usize, - /// Triangle SIMD packets of the indexed layout (vertex indices). - pub packets_indexed: usize, - /// Leaf primitive indices. - pub indices: usize, - /// Triangle records: vertex indices, ids and mask, 24 bytes each. - pub triangle_records: usize, - /// The shared vertex table, 12 bytes per vertex. - pub vertices: usize, - /// Per-vertex shading normals of the meshes that carry them, 12 bytes - /// per vertex. - pub vertex_normals: usize, - /// The per-geometry table (where each geometry's entries start). - pub geometry_tables: usize, - /// Packet lanes in total — a count, not bytes, so `total` ignores it. - pub lanes: usize, - /// Packet lanes holding a triangle; `lanes_filled / lanes` is the fill. - pub lanes_filled: usize, -} - -impl MemoryFootprint { - pub fn total(&self) -> usize { - self.prim_nodes - + self.instances - + self.cubic_spans - + self.bvh_nodes - + self.leaves - + self.packets - + self.packets_indexed - + self.indices - + self.triangle_records - + self.vertices - + self.vertex_normals - + self.geometry_tables - } -} - -/// Top-level primitives of a committed [`Scene`], split by kind. See -/// [`Scene::primitive_breakdown`]. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub struct PrimitiveBreakdown { - pub triangles: usize, - pub spheres: usize, - pub disks: usize, - pub cylinders: usize, - pub curve_segments: usize, - pub cubic_curve_spans: usize, - pub instances: usize, -} - -/// One authored cubic curve span — its own Bézier control points and end -/// radii, intersected analytically (`crate::curve::cubic_curve_intersect`) -/// rather than pre-flattened into several [`CurveSegment`]s. Round, same -/// as `RoundCurves` — this only changes how a span is stored and -/// intersected, not the surface it represents. -#[derive(Clone, Copy, Debug)] -pub struct CubicCurveSegment { - pub cp: [Vec3A; 4], - pub r0: f32, - pub r1: f32, -} - -/// A geometry to attach to a scene. The variants mirror Embree's geometry -/// types (the subset crust needs): triangle meshes, analytic spheres, -/// disks and open cylinders, round curves, and instances of another -/// committed scene. Instances nest: -/// an instanced scene may itself contain instances, and transforms, -/// normals and ray masks compose correctly through every level. -pub enum Geometry { - /// Unpadded `[f32; 3]` arrays, which is how the committed scene stores - /// them: a `Vec3A` is 16 bytes for 12 of data, and these arrays are - /// the bulk of a scene's memory. - TriangleMesh { - vertices: Vec<[f32; 3]>, - indices: Vec<[u32; 3]>, - /// Optional per-vertex shading normals; hits interpolate them by - /// the barycentrics (`SmoothTriangle` semantics). - normals: Option>, - }, - Sphere { - center: Vec3A, - radius: f32, - }, - /// A flat circular disk. `normal` names its front: hits report it as - /// the outward normal, so `RayHit::front_face` tells the two sides apart. - /// Need not be unit length; it is normalised at commit. - Disk { - center: Vec3A, - normal: Vec3A, - radius: f32, - }, - /// The side wall of a circular cylinder from `p0` to `p1` — open, with no - /// end caps. Hits report the radial outward normal. - Cylinder { - p0: Vec3A, - p1: Vec3A, - radius: f32, - }, - RoundCurves { - segments: Vec, - }, - /// Cubic curve spans, intersected as true curves instead of being - /// flattened to `RoundCurves` polylines — see [`CubicCurveSegment`]. - CubicCurves { - segments: Vec, - }, - /// Another committed scene placed by `transform` (local-to-world). - /// With `transform_end`, the placement interpolates linearly (per - /// matrix element) at the ray's shutter time — transform motion blur. - /// `transform` must be invertible. - Instance { - scene: Arc, - transform: Affine3A, - /// Boxed because it's `None` for the overwhelming majority of - /// instances (only motion-blurred placements set it): inline it - /// and every `Geometry` value — the enum is sized by its largest - /// variant — pays an extra 64 bytes it never uses. A scene with - /// millions of `PointInstancer` placements makes that the - /// dominant cost of the whole geometry table. - transform_end: Option>, - }, -} - -/// An intersection: distance, ray-facing normal (flipped to oppose the -/// ray, with `front_face` recording the original orientation), the hit -/// barycentrics where meaningful, and the IDs that let the application -/// map the hit back to its own data (materials, lights, …). -/// -/// For hits inside an [`Geometry::Instance`], `geom_id` is the -/// *instance's* id in the queried scene and `prim_id` the primitive index -/// within the instanced scene — the application maps per top-level -/// geometry. -/// -/// A curve hit also reports where along the curve it lies — `u` in `[0, 1]` -/// across its segment or cubic span — and `dpdu`, the curve's direction -/// there, unnormalised, in the queried scene's space (carried through every -/// instance by its local-to-world linear part, at the ray's time). Every -/// other hit's `dpdu` is zero. -#[derive(Clone, Copy, Debug)] -pub struct RayHit { - pub t: f32, - pub normal: Vec3A, - pub dpdu: Vec3A, - pub front_face: bool, - pub u: f32, - pub v: f32, - pub geom_id: u32, - pub prim_id: u32, -} - -/// Which `geom_id` a hit found inside an [`Geometry::Instance`] reports. -/// -/// By default an instance reports its *own* id and the inner ids are lost, -/// which is all a host that maps one material per top-level geometry needs. -/// A prototype of many parts wants more: to be placed as *one* instance -/// (so the BVH above it sees one box per placement, not one per part) and -/// still have a hit say which part it landed on. Embree answers with an -/// instance-id stack beside the inner `geomID`; this answers with one id, -/// computed as the hit passes back out through each instance level, which -/// keeps [`RayHit`] and the host's lookup a single index. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub enum InstanceHitId { - /// The instance's own `geom_id` in the scene it is attached to. - #[default] - Own, - /// A fixed id, whatever the inner scene reported. - As(u32), - /// `base` plus the id the inner scene reported, so an inner scene whose - /// hits already carry `0..n` maps onto `base..base + n` here. Nests: - /// each level adds its own base. [`SceneBuilder::commit`] panics if - /// `base` plus the largest id the inner scene can report would leave - /// the id space, rather than let a hit wrap onto another geometry. - Offset(u32), -} - -/// Accumulates geometries, then builds the acceleration structure once in -/// [`SceneBuilder::commit`] (Embree's `rtcCommitScene`). -#[derive(Default)] -pub struct SceneBuilder { - geoms: Vec<(Geometry, RayMask, InstanceHitId)>, -} - -impl SceneBuilder { - pub fn new() -> Self { - Self::default() - } - - /// Attaches a geometry visible to every ray category; returns its - /// `geom_id` (dense, starting at 0 — usable as a table index). - pub fn attach(&mut self, geometry: Geometry) -> u32 { - self.attach_masked(geometry, MASK_ALL) - } - - /// Attaches a geometry visible only to ray categories in `mask`. - pub fn attach_masked(&mut self, geometry: Geometry, mask: RayMask) -> u32 { - self.attach_labelled(geometry, mask, InstanceHitId::Own) - } - - /// Attaches an instance whose hits report `label` rather than the - /// instance's own id (see [`InstanceHitId`]). Returns the instance's own - /// `geom_id` all the same: it still occupies a slot. - /// - /// # Panics - /// If `label` is not [`InstanceHitId::Own`] and `geometry` is not an - /// instance — only an instance has inner hits to relabel. - pub fn attach_labelled( - &mut self, - geometry: Geometry, - mask: RayMask, - label: InstanceHitId, - ) -> u32 { - assert!( - label == InstanceHitId::Own || matches!(geometry, Geometry::Instance { .. }), - "only an instance can relabel its hits" - ); - self.geoms.push((geometry, mask, label)); - (self.geoms.len() - 1) as u32 - } - - /// Reserves capacity for `additional` more geometries. Purely a - /// performance/memory hint — callers that know an upcoming batch size - /// (a `PointInstancer` with N placements, say) should use it: without - /// it, growing a multi-million-entry `Vec` by repeated doubling both - /// re-copies everything so far at each doubling and can leave up to - /// ~2x the final size over-allocated. - pub fn reserve(&mut self, additional: usize) { - self.geoms.reserve(additional); - } - - /// Number of geometries attached so far. - pub fn count(&self) -> usize { - self.geoms.len() - } - - /// Replaces the geometry already attached at `id`, keeping its mask. - /// - /// For callers that must claim a `geom_id` before they can decide what - /// geometry belongs in it. The importer needs this: whether a mesh is - /// better placed by an instance or baked into world-space triangles - /// depends on how many times it turns out to be placed, which is not - /// known until the whole stage has been walked — but `geom_id`s are - /// handed out in traversal order and are the key the host's material - /// table is indexed by, so they cannot be assigned later. - /// - /// Attach a placeholder, keep the id, and fill it in here once the - /// decision is made. Only valid before [`SceneBuilder::commit`], which is - /// enforced by taking `&mut self`. - /// - /// # Panics - /// If `id` was never attached. - pub fn set_geometry(&mut self, id: u32, geometry: Geometry) { - self.geoms[id as usize].0 = geometry; - } - - /// The ray mask the geometry at `id` was attached with. - /// - /// # Panics - /// If `id` was never attached. - pub fn mask(&self, id: u32) -> RayMask { - self.geoms[id as usize].1 - } - - /// Replaces the ray mask of the geometry already attached at `id`, - /// keeping the geometry. The same deferred-decision escape hatch as - /// [`SceneBuilder::set_geometry`], for a visibility that is only known - /// once the whole input has been read. Only valid before - /// [`SceneBuilder::commit`]. - /// - /// # Panics - /// If `id` was never attached. - pub fn set_mask(&mut self, id: u32, mask: RayMask) { - self.geoms[id as usize].1 = mask; - } - - /// A geometry that expands to no primitives — the placeholder to pair - /// with [`SceneBuilder::set_geometry`]. Allocates nothing. - pub fn empty_geometry() -> Geometry { - Geometry::TriangleMesh { - vertices: Vec::new(), - indices: Vec::new(), - normals: None, - } - } - - /// How many primitives a geometry expands into. An upper bound: the - /// expansion skips degenerate entries (out-of-range indices, an empty - /// instanced scene), so the real count can be lower. - fn prim_upper_bound(geom: &Geometry) -> usize { - match geom { - Geometry::TriangleMesh { indices, .. } => indices.len(), - Geometry::RoundCurves { segments } => segments.len(), - Geometry::CubicCurves { segments } => segments.len(), - Geometry::Sphere { .. } - | Geometry::Disk { .. } - | Geometry::Cylinder { .. } - | Geometry::Instance { .. } => 1, - } - } - - /// Expands every geometry into primitives and builds the BVH with the - /// default [`CommitOptions`]. - #[must_use = "the committed scene is the only way to intersect it"] - pub fn commit(self) -> Scene { - self.commit_with(CommitOptions::default()) - } - - /// [`SceneBuilder::commit`] with the build choices made by the caller. - #[must_use = "the committed scene is the only way to intersect it"] - pub fn commit_with(self, options: CommitOptions) -> Scene { - let n_geoms = self.geoms.len() as u32; - // Size the primitive array exactly once, from the total the - // geometries will expand into, instead of letting per-geometry - // `reserve` calls grow it by doubling. With many geometries that - // doubling leaves up to ~2x over-allocated — and since a primitive - // node is 128 bytes, on a scene of a few hundred thousand baked - // triangles the waste was tens of MiB of genuinely committed memory - // (`MemoryFootprint` counts capacity, not length, which is why it - // showed up). The bound can only over-shoot by the number of - // degenerate primitives skipped below, normally zero. - let total: usize = self - .geoms - .iter() - .map(|(g, _, _)| Self::prim_upper_bound(g)) - .sum(); - let n_tris: usize = self - .geoms - .iter() - .map(|(g, _, _)| match g { - Geometry::TriangleMesh { indices, .. } => indices.len(), - _ => 0, - }) - .sum(); - let n_verts: usize = self - .geoms - .iter() - .map(|(g, _, _)| match g { - Geometry::TriangleMesh { vertices, .. } => vertices.len(), - _ => 0, - }) - .sum(); - assert!( - n_verts < u32::MAX as usize, - "{n_verts} vertices in one scene: the shared vertex table is indexed by u32" - ); - // Each non-triangle kind has its own array now; size each exactly. - let (mut n_inst, mut n_cubic) = (0usize, 0usize); - for (g, _, _) in &self.geoms { - match g { - Geometry::Instance { .. } => n_inst += 1, - Geometry::CubicCurves { segments } => n_cubic += segments.len(), - _ => {} - } - } - let n_nodes = total - n_tris - n_inst - n_cubic; - let mut input = Primitives { - tris: Vec::with_capacity(n_tris), - vertices: Vec::with_capacity(n_verts), - normals: Vec::new(), - geoms: Vec::with_capacity(self.geoms.len()), - prims: Vec::with_capacity(n_nodes), - instances: Vec::with_capacity(n_inst), - instance_bounds: Vec::with_capacity(n_inst), - cubics: Vec::with_capacity(n_cubic), - order: Vec::with_capacity(total - n_tris), - }; - let mut has_motion = false; - // Largest id a hit in this scene can report. Every geometry can - // report its own id; labels can report more (see below). - let mut max_hit_id = n_geoms.saturating_sub(1); - for (geom_id, (geom, mask, label)) in self.geoms.into_iter().enumerate() { - let geom_id = geom_id as u32; - let mut table = GeomTable { - vertex_base: input.vertices.len() as u32, - tri_base: input.tris.len() as u32, - ..GeomTable::default() - }; - match geom { - Geometry::TriangleMesh { - vertices, - indices, - normals, - } => { - // Normals are per vertex, parallel to the vertex run, - // and only when the array covers every vertex — a - // short one is ignored whole, as the per-triangle - // check it replaces ignored each triangle it fell - // short of. - let n = vertices.len(); - let vertex_base = table.vertex_base; - input.vertices.extend(vertices); - if let Some(mut ns) = normals.filter(|ns| ns.len() >= n) { - table.normal_base = input.normals.len() as u32; - ns.truncate(n); - input.normals.extend(ns); - } - for (prim_id, [i0, i1, i2]) in indices.into_iter().enumerate() { - let in_range = (i0 as usize) < n && (i1 as usize) < n && (i2 as usize) < n; - let v = if in_range { - [vertex_base + i0, vertex_base + i1, vertex_base + i2] - } else { - [DEGENERATE_VERTEX; 3] - }; - input.tris.push(TriangleRecord { - v, - geom_id, - prim_id: prim_id as u32, - mask: if in_range { mask } else { RayMask::NONE }, - }); - } - } - Geometry::Sphere { center, radius } => { - input.push_node(PrimNode::Sphere(SpherePrim { - center, - radius, - geom_id, - mask, - })); - } - Geometry::Disk { - center, - normal, - radius, - } => { - // Degenerate or non-finite: no front, no extent, or a - // position that would poison the bounds. - if !center.is_finite() - || !normal.is_finite() - || normal.length_squared() == 0.0 - || !radius.is_finite() - || radius <= 0.0 - { - continue; - } - input.push_node(PrimNode::Disk(DiskPrim { - center, - normal: normal.normalize(), - radius, - geom_id, - mask, - })); - } - Geometry::Cylinder { p0, p1, radius } => { - let length = (p1 - p0).length(); - if !p0.is_finite() - || !p1.is_finite() - || !length.is_finite() - || length <= 0.0 - || !radius.is_finite() - || radius <= 0.0 - { - continue; - } - input.push_node(PrimNode::Cylinder(CylinderPrim { - p0, - axis: (p1 - p0) / length, - length, - radius, - geom_id, - mask, - })); - } - Geometry::RoundCurves { segments } => { - let joints = - curve_joints(segments.iter().map(|s| (s.p0, s.p1, s.r0.min(s.r1)))); - for (prim_id, (s, joints)) in segments.into_iter().zip(joints).enumerate() { - input.push_node(PrimNode::Curve(CurvePrim { - p0: s.p0.to_array(), - p1: s.p1.to_array(), - r0: s.r0, - r1: s.r1, - geom_id, - prim_id: prim_id as u32, - mask, - joints, - })); - } - } - Geometry::CubicCurves { segments } => { - let joints = - curve_joints(segments.iter().map(|s| (s.cp[0], s.cp[3], s.r0.min(s.r1)))); - for (prim_id, (s, joints)) in segments.into_iter().zip(joints).enumerate() { - input.push_cubic(CubicCurvePrim { - cp: s.cp, - r0: s.r0, - r1: s.r1, - geom_id, - prim_id: prim_id as u32, - mask, - joints, - }); - } - } - Geometry::Instance { - scene, - transform, - transform_end, - } => { - let Some(inner_bounds) = scene.bounds() else { - continue; // empty instanced scene - }; - let w2l = transform.inverse(); - let bounds = match &transform_end { - Some(end) => AABB::surrounding_box( - transformed_aabb(&inner_bounds, &transform), - transformed_aabb(&inner_bounds, end), - ), - None => transformed_aabb(&inner_bounds, &transform), - }; - // A scene moves if this placement is blurred, or if the - // thing being placed already moves. The inner scene - // carries its own committed flag, so this stays O(1) per - // instance however deeply they nest. - has_motion |= transform_end.is_some() || scene.has_motion(); - let (geom_id, id_offset) = match label { - InstanceHitId::Own => (geom_id, NO_ID_OFFSET), - InstanceHitId::As(id) => { - assert!(id != crate::INVALID_ID, "hit id {id} is reserved"); - max_hit_id = max_hit_id.max(id); - (id, NO_ID_OFFSET) - } - InstanceHitId::Offset(base) => { - // Checked here, once per instance, so the hit - // path can add without a branch: an offset that - // could carry an inner id past the id space - // would otherwise wrap onto an unrelated - // geometry, and a host would shade it with that - // geometry's material. - let top = base - .checked_add(scene.max_hit_id) - .filter(|&top| top != crate::INVALID_ID) - .unwrap_or_else(|| { - panic!( - "id offset {base} + inner ids up to {} overflows the \ - geom_id space", - scene.max_hit_id - ) - }); - max_hit_id = max_hit_id.max(top); - (geom_id, base) - } - }; - input.push_instance( - InstancePrim { - scene, - w2l, - motion: transform_end.map(|end| { - Box::new(InstanceMotion { - l2w: transform, - l2w_end: *end, - }) - }), - geom_id, - id_offset, - mask, - }, - bounds, - ); - } - } - input.geoms.push(table); - } - let layout = match options.layout { - PacketLayout::Gathered | PacketLayout::Auto => crate::bvh::Layout::Gathered, - PacketLayout::Indexed => crate::bvh::Layout::Indexed, - }; - Scene { - bvh: Bvh::new(input, layout), - n_geoms, - has_motion, - max_hit_id, - } - } -} - -/// A committed, immutable scene. Queries are `&self` and thread-safe. -pub struct Scene { - bvh: Bvh, - n_geoms: u32, - has_motion: bool, - /// Largest `geom_id` a hit in this scene can report — a bound, computed - /// at commit, that lets an `Offset` label placing this scene be checked - /// once instead of on every hit. - max_hit_id: u32, -} - -impl Scene { - /// Closest hit in `(t_min, t_max)`, or `None` (Embree's - /// `rtcIntersect1`). - #[must_use] - pub fn intersect(&self, ray: &Ray, t_min: f32, t_max: f32) -> Option { - let hit = self.bvh.hit(ray, t_min, t_max)?; - let front_face = ray.dir.dot(hit.outward) < 0.0; - Some(RayHit { - t: hit.t, - normal: if front_face { - hit.outward - } else { - -hit.outward - }, - dpdu: hit.dpdu, - front_face, - u: hit.u, - v: hit.v, - geom_id: hit.geom_id, - prim_id: hit.prim_id, - }) - } - - /// Does the ray hit *anything* in `(t_min, t_max)`? Early-exit - /// traversal — the shadow-ray fast path (Embree's `rtcOccluded1`). - #[must_use] - pub fn occluded(&self, ray: &Ray, t_min: f32, t_max: f32) -> bool { - self.bvh.hit_any(ray, t_min, t_max) - } - - /// What the top-level instances `ids` are: `(geom_id, world bounds, - /// inner top-level primitive count, how many top-level instances share - /// the same inner scene)`. Diagnostic for a top level that will not - /// cull, paired with [`crate::traversal_stats::top_level_descents`]; - /// a linear scan, so ask once. - #[cfg(feature = "traversal-stats")] - pub fn describe_instances( - &self, - ids: &std::collections::HashSet, - ) -> Vec<(u32, AABB, usize, usize)> { - let mut sharing = std::collections::HashMap::<*const Scene, usize>::new(); - let mut found = Vec::new(); - for inst in self.bvh.instances() { - *sharing.entry(Arc::as_ptr(&inst.scene)).or_insert(0) += 1; - if ids.contains(&inst.geom_id) { - let bounds = inst - .approx_world_bounds() - .unwrap_or(AABB::new(Vec3A::ZERO, Vec3A::ZERO)); - found.push(( - inst.geom_id, - bounds, - Arc::as_ptr(&inst.scene), - inst.scene.primitive_count(), - )); - } - } - found - .into_iter() - .map(|(id, b, ptr, n)| (id, b, n, sharing[&ptr])) - .collect() - } - - /// World bounds of everything in the scene; `None` when empty. - pub fn bounds(&self) -> Option { - self.bvh.bounds() - } - - /// Number of attached geometries (`geom_id`s are `0..count`). - pub fn geometry_count(&self) -> u32 { - self.n_geoms - } - - /// Does anything in this scene move over the shutter interval — i.e. can - /// `ray.time` change what a query returns? - /// - /// Transform motion blur is the only thing that reads `ray.time` - /// (`InstancePrim::transforms_at`), so when this is `false` the shutter - /// coordinate is unobservable and a host need not sample it. That is - /// worth asking about: drawing one costs a full 4-dimensional - /// quasi-random sample, which was 4.2% of the render on a static scene. - /// - /// True if any instance carries an end-of-shutter transform, at any depth - /// of nesting (each level folds in its inner scene's answer at commit). - pub fn has_motion(&self) -> bool { - self.has_motion - } - - /// Number of primitives the geometries expanded into. - pub fn primitive_count(&self) -> usize { - self.bvh.prim_count() - } - - /// Top-level primitives split by kind, for reporting. Instances count - /// as one primitive each and are *not* descended into — the instanced - /// scene's own contents are its own `Scene`'s business, and a - /// prototype shared by a thousand placements would otherwise be - /// counted a thousand times. - pub fn primitive_breakdown(&self) -> PrimitiveBreakdown { - self.bvh.primitive_breakdown() - } - - /// Primitives actually resident in memory: like - /// [`Scene::primitive_breakdown`], but descending into instanced - /// scenes, counting each distinct prototype **once** however many - /// placements reference it. - /// - /// This is the count that tracks memory. `primitive_breakdown` says - /// what the top-level BVH traverses; this says what is stored. For an - /// instance-heavy scene the two differ enormously, and the gap is the - /// whole benefit of instancing. - pub fn unique_primitive_breakdown(&self) -> PrimitiveBreakdown { - let mut visited = std::collections::HashSet::new(); - let mut acc = PrimitiveBreakdown::default(); - self.accumulate_unique_into(&mut visited, &mut acc); - acc - } - - pub(crate) fn accumulate_unique_into( - &self, - visited: &mut std::collections::HashSet, - acc: &mut PrimitiveBreakdown, - ) { - self.bvh.accumulate_unique(visited, acc); - } - - /// How big this scene's top-level primitives are relative to the - /// scene itself: `(count, scene diagonal, mean prim diagonal, max prim - /// diagonal)`. - /// - /// Diagnostic for a BVH that will not cull. A hierarchy can only - /// separate primitives whose bounds are small against the whole; when - /// the mean ratio approaches 1 every box covers everything, no split - /// can divide them, and traversal degenerates to a linear scan however - /// good the builder is. - pub fn primitive_extents(&self) -> (usize, f32, f32, f32) { - let scene_diag = self - .bvh - .bounds() - .map(|b| (b.maximum - b.minimum).length()) - .unwrap_or(0.0); - let (n, sum, max) = self.bvh.primitive_extent_sum(); - let mean = if n == 0 { 0.0 } else { sum / n as f32 }; - (n, scene_diag, mean, max) - } - - /// The three vertices of triangle `prim_id` of the triangle mesh - /// `geom_id`, in this scene's own space — local space for a scene that - /// is placed through instances. `None` when `geom_id` is not a - /// triangle mesh of this scene, `prim_id` is past its triangles, or - /// the triangle's attached indices were out of range. - /// - /// What lets an application derive per-hit quantities (a tangent - /// frame, a texture density) from the geometry it attached instead of - /// storing them per triangle. - pub fn triangle_vertices(&self, geom_id: u32, prim_id: u32) -> Option<[Vec3A; 3]> { - self.bvh.triangle_vertices(geom_id, prim_id) - } - - /// Exact resident bytes of this scene and every distinct scene it - /// instances — see [`MemoryFootprint`]. - pub fn memory_footprint(&self) -> MemoryFootprint { - let mut visited = std::collections::HashSet::new(); - let mut acc = MemoryFootprint::default(); - self.accumulate_footprint_into(&mut visited, &mut acc); - acc - } - - pub(crate) fn accumulate_footprint_into( - &self, - visited: &mut std::collections::HashSet, - acc: &mut MemoryFootprint, - ) { - self.bvh.accumulate_footprint(visited, acc); - } - - /// Internal closest-hit that keeps the *outward* (unoriented) normal, - /// so instance transforms can map it without re-deriving orientation. - pub(crate) fn intersect_outward(&self, ray: &Ray, t_min: f32, t_max: f32) -> Option { - self.bvh.hit(ray, t_min, t_max) - } -} - -/// Which ends of each segment of a curve batch continue into its neighbour -/// in the batch: segment `i`'s end and segment `i + 1`'s start, where they -/// meet. A strand is attached as consecutive segments (or spans), and a cap -/// where two of them meet is buried in the next one's body except on the -/// outside of a bend, so a ray passing out of tubes must not meet it. Where -/// they meet is decided within float rounding of the coordinates (a B-spline -/// converted to Bézier form shares its span ends only to the last ulps) and -/// a thousandth of the radius: two strands whose ends merely touch are not -/// one strand to any visible degree. -fn curve_joints(ends: impl Iterator) -> Vec { - let ends: Vec<_> = ends.collect(); - let meet = |a: Vec3A, b: Vec3A, r: f32| { - let scale = a.abs().max(b.abs()).max_element(); - (a - b).abs().max_element() <= 4.0 * f32::EPSILON * scale + 1e-3 * r - }; - let mut joints = vec![0u8; ends.len()]; - for i in 1..ends.len() { - let ((_, prev_end, r0), (start, _, r1)) = (ends[i - 1], ends[i]); - if meet(prev_end, start, r0.min(r1)) { - joints[i - 1] |= crate::curve::JOINED_END; - joints[i] |= crate::curve::JOINED_START; - } - } - joints -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::ray::{MASK_CAMERA, MASK_SHADOW}; - use glam::Mat4; - - fn unit_sphere_scene() -> Arc { - let mut b = SceneBuilder::new(); - b.attach(Geometry::Sphere { - center: Vec3A::ZERO, - radius: 1.0, - }); - Arc::new(b.commit()) - } - - /// A disk is hit inside its radius and nowhere else, from either side, - /// and `front_face` names the side its `normal` points to. - #[test] - fn disk_is_flat_round_and_knows_its_front() { - let mut b = SceneBuilder::new(); - let id = b.attach(Geometry::Disk { - center: Vec3A::new(0.0, 0.0, 2.0), - normal: Vec3A::new(0.0, 0.0, -3.0), // unnormalised on purpose - radius: 1.0, - }); - let s = b.commit(); - - let from_front = s - .intersect( - &Ray::new(Vec3A::new(0.5, 0.5, 0.0), Vec3A::Z), - 0.0, - f32::INFINITY, - ) - .expect("inside the radius"); - assert_eq!(from_front.geom_id, id); - assert!((from_front.t - 2.0).abs() < 1e-6); - assert!( - from_front.front_face, - "the ray arrives on the -Z (front) side" - ); - assert!(from_front.normal.abs_diff_eq(-Vec3A::Z, 1e-6)); - - let from_back = s - .intersect( - &Ray::new(Vec3A::new(0.0, 0.0, 5.0), -Vec3A::Z), - 0.0, - f32::INFINITY, - ) - .expect("a disk is visible from behind too"); - assert!(!from_back.front_face); - - // Just outside the radius (0.72² + 0.72² > 1) misses; parallel misses. - assert!( - s.intersect( - &Ray::new(Vec3A::new(0.72, 0.72, 0.0), Vec3A::Z), - 0.0, - f32::INFINITY - ) - .is_none() - ); - assert!( - s.intersect( - &Ray::new(Vec3A::new(-3.0, 0.0, 2.0), Vec3A::X), - 0.0, - f32::INFINITY - ) - .is_none() - ); - - // Bounds are exact in the plane and padded across it. - let bb = s.bounds().unwrap(); - assert!((bb.maximum.x - 1.0).abs() < 1e-6 && (bb.minimum.y + 1.0).abs() < 1e-6); - assert!(bb.maximum.z > bb.minimum.z); - } - - /// Degenerate or non-finite disks and cylinders are skipped at commit - /// rather than handed to the BVH, where one infinite bound would poison - /// every box above it. - #[test] - fn invalid_disks_and_cylinders_are_skipped() { - let mut b = SceneBuilder::new(); - for g in [ - Geometry::Disk { - center: Vec3A::ZERO, - normal: Vec3A::ZERO, - radius: 1.0, - }, - Geometry::Disk { - center: Vec3A::ZERO, - normal: Vec3A::Z, - radius: -1.0, - }, - Geometry::Disk { - center: Vec3A::ZERO, - normal: Vec3A::Z, - radius: f32::INFINITY, - }, - Geometry::Disk { - center: Vec3A::splat(f32::NAN), - normal: Vec3A::Z, - radius: 1.0, - }, - Geometry::Cylinder { - p0: Vec3A::ZERO, - p1: Vec3A::ZERO, - radius: 1.0, - }, - Geometry::Cylinder { - p0: Vec3A::ZERO, - p1: Vec3A::X, - radius: 0.0, - }, - Geometry::Cylinder { - p0: Vec3A::ZERO, - p1: Vec3A::splat(f32::INFINITY), - radius: 1.0, - }, - ] { - b.attach(g); - } - let s = b.commit(); - assert_eq!( - s.primitive_breakdown().disks + s.primitive_breakdown().cylinders, - 0 - ); - assert!(s.bounds().is_none()); - } - - /// An open tube: the wall is hit from outside with an outward normal, - /// from inside as a back face, and the ends are open. - #[test] - fn cylinder_is_an_open_tube() { - let mut b = SceneBuilder::new(); - b.attach(Geometry::Cylinder { - p0: Vec3A::new(-1.0, 0.0, 0.0), - p1: Vec3A::new(1.0, 0.0, 0.0), - radius: 0.5, - }); - let s = b.commit(); - - let outside = s - .intersect( - &Ray::new(Vec3A::new(0.3, 0.0, -4.0), Vec3A::Z), - 0.0, - f32::INFINITY, - ) - .expect("the wall"); - assert!((outside.t - 3.5).abs() < 1e-5); - assert!(outside.front_face); - assert!(outside.normal.abs_diff_eq(-Vec3A::Z, 1e-5)); - - let inside = s - .intersect( - &Ray::new(Vec3A::new(0.3, 0.0, 0.0), Vec3A::Y), - 0.0, - f32::INFINITY, - ) - .expect("the wall, from within"); - assert!((inside.t - 0.5).abs() < 1e-5); - assert!(!inside.front_face, "the inside of a tube is its back face"); - - // Down the axis there are no caps to hit; past the end, no wall. - assert!( - s.intersect( - &Ray::new(Vec3A::new(-5.0, 0.1, 0.0), Vec3A::X), - 0.0, - f32::INFINITY - ) - .is_none() - ); - assert!( - s.intersect( - &Ray::new(Vec3A::new(1.2, 0.0, -4.0), Vec3A::Z), - 0.0, - f32::INFINITY - ) - .is_none() - ); - // An oblique ray entering past the end exits through the wall. - let oblique = s - .intersect( - &Ray::new(Vec3A::new(1.5, 0.0, 0.0), Vec3A::new(-1.0, 0.0, 1.0)), - 0.0, - f32::INFINITY, - ) - .expect("exits through the wall"); - assert!(!oblique.front_face); - - let bb = s.bounds().unwrap(); - assert!(bb.minimum.abs_diff_eq(Vec3A::new(-1.0, -0.5, -0.5), 1e-6)); - assert!(bb.maximum.abs_diff_eq(Vec3A::new(1.0, 0.5, 0.5), 1e-6)); - assert_eq!(s.primitive_breakdown().cylinders, 1); - } - - /// Placed through an instance, a disk under a non-uniform scale is an - /// ellipse — the path the importer takes for a squashed `DiskLight`. - #[test] - fn instanced_disk_becomes_an_ellipse() { - let mut inner = SceneBuilder::new(); - inner.attach(Geometry::Disk { - center: Vec3A::ZERO, - normal: -Vec3A::Z, - radius: 1.0, - }); - let inner = Arc::new(inner.commit()); - let mut b = SceneBuilder::new(); - b.attach(Geometry::Instance { - scene: inner, - transform: Affine3A::from_scale(glam::Vec3::new(3.0, 1.0, 1.0)), - transform_end: None, - }); - let s = b.commit(); - let hit = |x: f32| s.intersect(&Ray::new(Vec3A::new(x, 0.0, -1.0), Vec3A::Z), 0.0, 10.0); - assert!(hit(2.9).is_some_and(|h| h.front_face)); - assert!(hit(3.1).is_none()); - } - - /// A reserved slot must keep its `geom_id` (so ids stay dense and every - /// later attach is unperturbed) and contribute nothing until filled. - #[test] - fn reserved_slots_keep_ids_dense_and_stay_invisible() { - let mut b = SceneBuilder::new(); - let a = b.attach(Geometry::Sphere { - center: Vec3A::new(-5.0, 0.0, 0.0), - radius: 1.0, - }); - let placeholder = b.attach(SceneBuilder::empty_geometry()); - let c = b.attach(Geometry::Sphere { - center: Vec3A::new(5.0, 0.0, 0.0), - radius: 1.0, - }); - assert_eq!((a, placeholder, c), (0, 1, 2), "ids stay dense"); - - // Never filled in: the two spheres are all there is. - let scene = b.commit(); - assert_eq!(scene.geometry_count(), 3); - assert_eq!(scene.primitive_count(), 2); - - // Filling it in later puts real geometry under the id it claimed. - let mut b = SceneBuilder::new(); - b.attach(Geometry::Sphere { - center: Vec3A::new(-5.0, 0.0, 0.0), - radius: 1.0, - }); - let slot = b.attach(SceneBuilder::empty_geometry()); - b.set_geometry( - slot, - Geometry::Sphere { - center: Vec3A::ZERO, - radius: 1.0, - }, - ); - let scene = b.commit(); - assert_eq!(scene.primitive_count(), 2); - let hit = scene - .intersect( - &Ray::new(Vec3A::new(0.0, 0.0, -8.0), Vec3A::Z), - 1e-4, - f32::MAX, - ) - .expect("the filled-in sphere is hit"); - assert_eq!(hit.geom_id, slot, "and reports the id it reserved"); - } - - #[test] - fn ids_map_back_to_geometries() { - let mut b = SceneBuilder::new(); - let ball = b.attach(Geometry::Sphere { - center: Vec3A::new(-3.0, 0.0, 0.0), - radius: 1.0, - }); - let quad = b.attach(Geometry::TriangleMesh { - vertices: vec![ - [2.0, -1.0, -1.0], - [2.0, -1.0, 1.0], - [2.0, 1.0, 1.0], - [2.0, 1.0, -1.0], - ], - indices: vec![[0, 1, 2], [0, 2, 3]], - normals: None, - }); - let scene = b.commit(); - assert_eq!(scene.geometry_count(), 2); - assert_eq!(scene.primitive_count(), 3); - - let hit_ball = scene - .intersect( - &Ray::new(Vec3A::new(-3.0, 0.0, -5.0), Vec3A::Z), - 0.001, - f32::INFINITY, - ) - .expect("ball hit"); - assert_eq!(hit_ball.geom_id, ball); - - // Aim at the second triangle of the quad (upper-left half). - let hit_quad = scene - .intersect( - &Ray::new(Vec3A::new(0.0, 0.5, -0.5), Vec3A::X), - 0.001, - f32::INFINITY, - ) - .expect("quad hit"); - assert_eq!(hit_quad.geom_id, quad); - assert_eq!(hit_quad.prim_id, 1); - } - - #[test] - fn front_face_semantics_match_ray_side() { - let scene = unit_sphere_scene(); - let outside = scene - .intersect( - &Ray::new(Vec3A::new(0.0, 0.0, -5.0), Vec3A::Z), - 0.001, - 100.0, - ) - .expect("outside hit"); - assert!(outside.front_face); - assert!(outside.normal.abs_diff_eq(-Vec3A::Z, 1e-4)); - - // From inside the sphere the normal flips toward the origin. - let inside = scene - .intersect(&Ray::new(Vec3A::ZERO, Vec3A::Z), 0.001, 100.0) - .expect("inside hit"); - assert!(!inside.front_face); - assert!(inside.normal.abs_diff_eq(-Vec3A::Z, 1e-4)); - } - - /// Does the kernel already nest instances? An `Instance` holds an - /// `Arc`, and nothing stops that scene from containing - /// instances of its own — this asks whether the recursion actually - /// works end to end, or only looks like it should. - #[test] - fn motion_flag_reports_a_static_scene_as_static() { - let mut b = SceneBuilder::new(); - b.attach(Geometry::Sphere { - center: Vec3A::ZERO, - radius: 1.0, - }); - assert!(!b.commit().has_motion()); - - // A *static* placement of static geometry is still static. - let mut b = SceneBuilder::new(); - b.attach(Geometry::Instance { - scene: unit_sphere_scene(), - transform: Affine3A::from_translation(glam::Vec3::new(3.0, 0.0, 0.0)), - transform_end: None, - }); - assert!(!b.commit().has_motion()); - } - - /// The way `has_motion` can be wrong that matters: motion authored on an - /// *inner* instance, placed by an outer one that does not itself move. - /// If the flag only looked at its own `transform_end`, the renderer would - /// stop sampling the shutter and silently drop the blur. - #[test] - fn motion_flag_propagates_through_nesting() { - let leaf = unit_sphere_scene(); - - // Level 1: the sphere streaks along X over the shutter. - let mut mid = SceneBuilder::new(); - mid.attach(Geometry::Instance { - scene: leaf, - transform: Affine3A::IDENTITY, - transform_end: Some(Box::new(Affine3A::from_translation(glam::Vec3::new( - 4.0, 0.0, 0.0, - )))), - }); - let mid = Arc::new(mid.commit()); - assert!(mid.has_motion(), "the level that authored the motion"); - - // Level 2: a static placement of that moving scene. Still moving. - let mut root = SceneBuilder::new(); - root.attach(Geometry::Instance { - scene: Arc::clone(&mid), - transform: Affine3A::from_translation(glam::Vec3::new(0.0, 7.0, 0.0)), - transform_end: None, - }); - assert!( - root.commit().has_motion(), - "a static placement of moving geometry still moves" - ); - } - - #[test] - fn instances_nest() { - // Level 0: a unit sphere at the origin. - let leaf = unit_sphere_scene(); - - // Level 1: two spheres, at local x = ±2. - let mut mid = SceneBuilder::new(); - for x in [-2.0f32, 2.0] { - mid.attach(Geometry::Instance { - scene: Arc::clone(&leaf), - transform: Affine3A::from_translation(glam::Vec3::new(x, 0.0, 0.0)), - transform_end: None, - }); - } - let mid = Arc::new(mid.commit()); - - // Level 2: two copies of that pair, at world y = ±5. Four spheres - // in total, from one copy of the sphere's geometry. - let mut root = SceneBuilder::new(); - for y in [-5.0f32, 5.0] { - root.attach(Geometry::Instance { - scene: Arc::clone(&mid), - transform: Affine3A::from_translation(glam::Vec3::new(0.0, y, 0.0)), - transform_end: None, - }); - } - let scene = root.commit(); - - // Two top-level primitives hold four spheres. - assert_eq!(scene.primitive_count(), 2); - - for (x, y) in [(-2.0f32, -5.0f32), (2.0, -5.0), (-2.0, 5.0), (2.0, 5.0)] { - let ray = Ray::new(Vec3A::new(x, y, -8.0), Vec3A::Z); - let hit = scene - .intersect(&ray, 0.001, 100.0) - .unwrap_or_else(|| panic!("nested sphere at ({x}, {y}) was missed")); - assert!( - (hit.t - 7.0).abs() < 1e-3, - "nested sphere at ({x}, {y}): t = {} (want 7)", - hit.t - ); - assert!( - hit.normal.abs_diff_eq(-Vec3A::Z, 1e-4), - "nested normal wrong at ({x}, {y}): {:?}", - hit.normal - ); - assert!(scene.occluded(&ray, 0.001, 100.0)); - } - - // And nothing where the spheres are not. - assert!( - scene - .intersect( - &Ray::new(Vec3A::new(0.0, 0.0, -8.0), Vec3A::Z), - 0.001, - 100.0 - ) - .is_none(), - "hit between the nested spheres" - ); - } - - /// The reporting counts: the top-level view sees instances, the unique - /// view descends but counts a shared prototype only once — the whole - /// point being that four placed spheres cost one sphere of memory. - #[test] - fn unique_breakdown_counts_shared_prototypes_once() { - let leaf = unit_sphere_scene(); - let mut mid = SceneBuilder::new(); - for x in [-2.0f32, 2.0] { - mid.attach(Geometry::Instance { - scene: Arc::clone(&leaf), - transform: Affine3A::from_translation(glam::Vec3::new(x, 0.0, 0.0)), - transform_end: None, - }); - } - let mid = Arc::new(mid.commit()); - let mut root = SceneBuilder::new(); - for y in [-5.0f32, 5.0] { - root.attach(Geometry::Instance { - scene: Arc::clone(&mid), - transform: Affine3A::from_translation(glam::Vec3::new(0.0, y, 0.0)), - transform_end: None, - }); - } - let scene = root.commit(); - - // Top level: the two outer placements, no spheres visible yet. - let top = scene.primitive_breakdown(); - assert_eq!(top.instances, 2); - assert_eq!(top.spheres, 0); - - // Unique: descends both levels, but `mid` is one Arc shared by two - // placements and `leaf` one Arc shared by two more — so exactly one - // sphere is resident, reached through 2 + 2 instance primitives. - let unique = scene.unique_primitive_breakdown(); - assert_eq!(unique.spheres, 1, "shared prototype counted more than once"); - assert_eq!(unique.instances, 4); - } - - /// Nested instances must compose transforms in the right order, and - /// map normals back through both levels. A rotation at the outer level - /// and a non-uniform scale at the inner level do not commute, so this - /// fails loudly if the composition is inverted. - #[test] - fn nested_instances_compose_transforms_and_normals() { - // Inner: a unit sphere squashed to an ellipsoid by the mid level. - let leaf = unit_sphere_scene(); - let mut mid = SceneBuilder::new(); - mid.attach(Geometry::Instance { - scene: leaf, - // 2x along local X only. - transform: Affine3A::from_scale(glam::Vec3::new(2.0, 1.0, 1.0)), - transform_end: None, - }); - let mid = Arc::new(mid.commit()); - - // Outer: rotate that ellipsoid 90 degrees about Z, so its long - // axis ends up along world Y. - let mut root = SceneBuilder::new(); - root.attach(Geometry::Instance { - scene: mid, - transform: Affine3A::from_rotation_z(std::f32::consts::FRAC_PI_2), - transform_end: None, - }); - let scene = root.commit(); - - // Long axis is now Y: a ray down the Y axis meets the surface at - // |y| = 2, while one down X meets it at |x| = 1. - let along_y = scene - .intersect( - &Ray::new(Vec3A::new(0.0, -8.0, 0.0), Vec3A::Y), - 0.001, - 100.0, - ) - .expect("ray along Y must hit the rotated ellipsoid"); - assert!( - (along_y.t - 6.0).abs() < 1e-3, - "long axis is not along Y: t = {} (want 6)", - along_y.t - ); - let along_x = scene - .intersect( - &Ray::new(Vec3A::new(-8.0, 0.0, 0.0), Vec3A::X), - 0.001, - 100.0, - ) - .expect("ray along X must hit the rotated ellipsoid"); - assert!( - (along_x.t - 7.0).abs() < 1e-3, - "short axis is not along X: t = {} (want 7)", - along_x.t - ); - - // The normal at the Y pole points back down -Y; an inverse - // transpose applied at only one level would tilt it. - assert!( - along_y.normal.abs_diff_eq(-Vec3A::Y, 1e-4), - "nested normal not mapped through both levels: {:?}", - along_y.normal - ); - } - - /// A ray mask must gate at every level of nesting: hiding the outer - /// instance hides everything beneath it. - #[test] - fn nested_instances_respect_masks_at_each_level() { - let leaf = unit_sphere_scene(); - let mut mid = SceneBuilder::new(); - mid.attach_masked( - Geometry::Instance { - scene: leaf, - transform: Affine3A::IDENTITY, - transform_end: None, - }, - MASK_SHADOW, - ); - let mid = Arc::new(mid.commit()); - - let mut root = SceneBuilder::new(); - root.attach_masked( - Geometry::Instance { - scene: mid, - transform: Affine3A::IDENTITY, - transform_end: None, - }, - MASK_SHADOW | MASK_CAMERA, - ); - let scene = root.commit(); - - let ray = Ray::new(Vec3A::new(0.0, 0.0, -5.0), Vec3A::Z); - // The inner level only admits shadow rays, so a camera ray is - // rejected there even though the outer level would allow it. - assert!( - scene - .intersect(&ray.with_mask(MASK_CAMERA), 0.001, 100.0) - .is_none() - ); - assert!( - scene - .intersect(&ray.with_mask(MASK_SHADOW), 0.001, 100.0) - .is_some() - ); - } - - #[test] - fn masks_filter_by_ray_category() { - let mut b = SceneBuilder::new(); - b.attach_masked( - Geometry::Sphere { - center: Vec3A::ZERO, - radius: 1.0, - }, - MASK_SHADOW, - ); - let scene = b.commit(); - let ray = Ray::new(Vec3A::new(0.0, 0.0, -5.0), Vec3A::Z); - assert!( - scene - .intersect(&ray.with_mask(MASK_CAMERA), 0.001, 100.0) - .is_none() - ); - assert!( - scene - .intersect(&ray.with_mask(MASK_SHADOW), 0.001, 100.0) - .is_some() - ); - assert!(!scene.occluded(&ray.with_mask(MASK_CAMERA), 0.001, 100.0)); - assert!(scene.occluded(&ray.with_mask(MASK_SHADOW), 0.001, 100.0)); - } - - /// A mask replaced after attaching is the one the committed scene - /// filters by, exactly as if it had been attached with it. - #[test] - fn set_mask_replaces_the_attached_mask() { - let mut b = SceneBuilder::new(); - let id = b.attach(Geometry::Sphere { - center: Vec3A::ZERO, - radius: 1.0, - }); - assert_eq!(b.mask(id), MASK_ALL); - b.set_mask(id, MASK_CAMERA); - assert_eq!(b.mask(id), MASK_CAMERA); - let scene = b.commit(); - let ray = Ray::new(Vec3A::new(0.0, 0.0, -5.0), Vec3A::Z); - assert!( - scene - .intersect(&ray.with_mask(MASK_CAMERA), 0.001, 100.0) - .is_some() - ); - assert!(!scene.occluded(&ray.with_mask(MASK_SHADOW), 0.001, 100.0)); - } - - #[test] - fn smooth_normals_interpolate() { - // One triangle with vertex normals fanned outward; the hit normal - // at an interior point must be a blend, not the face normal. - let mut b = SceneBuilder::new(); - b.attach(Geometry::TriangleMesh { - vertices: vec![[-1.0, -1.0, 2.0], [1.0, -1.0, 2.0], [0.0, 1.0, 2.0]], - indices: vec![[0, 1, 2]], - normals: Some(vec![ - Vec3A::new(-0.5, 0.0, -1.0).normalize().to_array(), - Vec3A::new(0.5, 0.0, -1.0).normalize().to_array(), - Vec3A::new(0.0, 0.5, -1.0).normalize().to_array(), - ]), - }); - let scene = b.commit(); - // Straight at the v1 corner region: x > 0 → normal tilts +x. - let hit = scene - .intersect( - &Ray::new(Vec3A::new(0.6, -0.7, 0.0), Vec3A::Z), - 0.001, - 100.0, - ) - .expect("hit"); - assert!( - hit.normal.x > 0.1, - "normal not interpolated: {:?}", - hit.normal - ); - assert!(hit.normal.z < 0.0); - } - - #[test] - fn translated_instance_matches_baked() { - let mut b = SceneBuilder::new(); - b.attach(Geometry::Instance { - scene: unit_sphere_scene(), - transform: Affine3A::from_translation(glam::Vec3::new(3.0, 0.0, 0.0)), - transform_end: None, - }); - let scene = b.commit(); - let ray = Ray::new(Vec3A::new(3.0, 0.0, -5.0), Vec3A::Z); - let hit = scene.intersect(&ray, 0.001, f32::INFINITY).expect("hit"); - assert!((hit.t - 4.0).abs() < 1e-4); - assert!(hit.normal.abs_diff_eq(-Vec3A::Z, 1e-4)); - assert!(scene.occluded(&ray, 0.001, f32::INFINITY)); - assert!(!scene.occluded(&ray, 0.001, 3.9)); - } - - #[test] - fn nonuniform_scale_transforms_normals_correctly() { - // Sphere scaled 2x in X: probe an oblique point where the naive - // (non inverse-transpose) normal mapping would be wrong. - let mut b = SceneBuilder::new(); - b.attach(Geometry::Instance { - scene: unit_sphere_scene(), - transform: Affine3A::from_scale(glam::Vec3::new(2.0, 1.0, 1.0)), - transform_end: None, - }); - let scene = b.commit(); - // Hit the ellipsoid straight down above x=1 (local x=0.5). - let hit = scene - .intersect( - &Ray::new(Vec3A::new(1.0, 5.0, 0.0), -Vec3A::Y), - 0.001, - f32::INFINITY, - ) - .expect("hit"); - // Implicit ellipsoid (x/2)^2 + y^2 + z^2 = 1: gradient at - // (1, sqrt(3)/2, 0) is proportional to (0.5, sqrt(3), 0). - let expected = Vec3A::new(0.5, 3.0f32.sqrt(), 0.0).normalize(); - assert!( - hit.normal.abs_diff_eq(expected, 1e-3), - "normal {:?} != expected {:?}", - hit.normal, - expected - ); - } - - #[test] - fn rotated_instance_hits_where_baked_triangle_would() { - let mut inner = SceneBuilder::new(); - inner.attach(Geometry::TriangleMesh { - vertices: vec![[-1.0, -1.0, 0.0], [1.0, -1.0, 0.0], [0.0, 1.0, 0.0]], - indices: vec![[0, 1, 2]], - normals: None, - }); - let mut b = SceneBuilder::new(); - b.attach(Geometry::Instance { - scene: Arc::new(inner.commit()), - transform: Affine3A::from_mat4( - Mat4::from_rotation_y(std::f32::consts::FRAC_PI_2) - * Mat4::from_translation(glam::Vec3::new(0.0, 0.0, 2.0)), - ), - transform_end: None, - }); - let scene = b.commit(); - // Local (0,0,2) maps to world (2,0,0); triangle now faces +X. - let hit = scene - .intersect( - &Ray::new(Vec3A::new(5.0, 0.0, 0.0), -Vec3A::X), - 0.001, - f32::INFINITY, - ) - .expect("hit"); - assert!((hit.t - 3.0).abs() < 1e-4); - assert!(hit.normal.abs_diff_eq(Vec3A::X, 1e-4)); - } - - #[test] - fn motion_blur_interpolates_position() { - let mut b = SceneBuilder::new(); - b.attach(Geometry::Instance { - scene: unit_sphere_scene(), - transform: Affine3A::IDENTITY, - transform_end: Some(Box::new(Affine3A::from_translation(glam::Vec3::new( - 4.0, 0.0, 0.0, - )))), - }); - let scene = b.commit(); - // At time 0 the sphere is at the origin... - let r0 = Ray::new(Vec3A::new(0.0, 0.0, -5.0), Vec3A::Z).with_time(0.0); - assert!(scene.intersect(&r0, 0.001, f32::INFINITY).is_some()); - // ...at time 1 it has moved to x=4... - let r1 = Ray::new(Vec3A::new(4.0, 0.0, -5.0), Vec3A::Z).with_time(1.0); - assert!(scene.intersect(&r1, 0.001, f32::INFINITY).is_some()); - let r1_origin = Ray::new(Vec3A::new(0.0, 0.0, -5.0), Vec3A::Z).with_time(1.0); - assert!(scene.intersect(&r1_origin, 0.001, f32::INFINITY).is_none()); - // ...and at time 0.5 it is halfway. - let rh = Ray::new(Vec3A::new(2.0, 0.0, -5.0), Vec3A::Z).with_time(0.5); - let hit = scene - .intersect(&rh, 0.001, f32::INFINITY) - .expect("halfway hit"); - assert!((hit.t - 4.0).abs() < 1e-4); - // The shutter-union bounding box covers both endpoints. - let bb = scene.bounds().unwrap(); - assert!(bb.minimum.x <= -1.0 && bb.maximum.x >= 5.0); - } - - /// A unit sphere at `x` in a one-geometry scene: a stand-in "part". - fn part_at(x: f32) -> Arc { - let mut b = SceneBuilder::new(); - b.attach(Geometry::Sphere { - center: Vec3A::new(x, 0.0, 0.0), - radius: 0.5, - }); - Arc::new(b.commit()) - } - - fn placed(scene: &Arc, z: f32) -> Geometry { - Geometry::Instance { - scene: scene.clone(), - transform: Affine3A::from_translation(glam::Vec3::new(0.0, 0.0, z)), - transform_end: None, - } - } - - /// The id a ray down +Z at `x` reports, if it hits. - fn id_at(scene: &Scene, x: f32) -> Option { - scene - .intersect( - &Ray::new(Vec3A::new(x, 0.0, -10.0), Vec3A::Z), - 0.001, - f32::INFINITY, - ) - .map(|h| h.geom_id) - } - - /// The shape the importer builds for a many-part prototype: parts - /// labelled `As(k)` inside a group, the group placed several times under - /// `Offset(0)` inside a scatter, and the scatter attached at the top - /// level under `Offset(base)`. Every hit must come out as `base + k`, - /// whichever placement it landed in. - #[test] - fn labelled_instances_compose_offsets_through_nesting() { - let mut group = SceneBuilder::new(); - for k in 0..3u32 { - // Attached out of order so a part's own id is never its label. - group.attach_labelled( - placed(&part_at(k as f32 * 2.0), 0.0), - MASK_ALL, - InstanceHitId::As(2 - k), - ); - } - let group = Arc::new(group.commit()); - - let mut scatter = SceneBuilder::new(); - for z in [0.0, 5.0] { - scatter.attach_labelled(placed(&group, z), MASK_ALL, InstanceHitId::Offset(0)); - } - let scatter = Arc::new(scatter.commit()); - - let mut top = SceneBuilder::new(); - let other = top.attach(Geometry::Sphere { - center: Vec3A::new(-5.0, 0.0, 0.0), - radius: 0.5, - }); - let base = 10; - let own = top.attach_labelled(placed(&scatter, 0.0), MASK_ALL, InstanceHitId::Offset(base)); - let top = top.commit(); - - assert_eq!(own, 1, "a labelled instance still takes its own slot"); - assert_eq!(id_at(&top, -5.0), Some(other)); - assert_eq!(id_at(&top, 0.0), Some(base + 2)); - assert_eq!(id_at(&top, 2.0), Some(base + 1)); - assert_eq!(id_at(&top, 4.0), Some(base)); - assert_eq!(id_at(&top, 1.0), None); - } - - /// `As` overrides the inner id outright, and `Own` still reports the - /// instance's slot even when what it places forwards. - #[test] - fn as_and_own_labels_override_inner_ids() { - let mut inner = SceneBuilder::new(); - inner.attach_labelled( - placed(&part_at(0.0), 0.0), - MASK_ALL, - InstanceHitId::Offset(7), - ); - let inner = Arc::new(inner.commit()); - - let mut b = SceneBuilder::new(); - let own = b.attach(placed(&inner, 0.0)); - let scene = b.commit(); - assert_eq!(id_at(&scene, 0.0), Some(own)); - - let mut b = SceneBuilder::new(); - b.attach_labelled(placed(&inner, 0.0), MASK_ALL, InstanceHitId::As(42)); - assert_eq!(id_at(&b.commit(), 0.0), Some(42)); - } - - /// An offset that could carry an inner id past the id space is refused - /// at commit: on the hit path it would wrap onto an unrelated id, and a - /// host would shade the hit with that geometry's material. - #[test] - #[should_panic(expected = "overflows the geom_id space")] - fn an_offset_that_could_overflow_is_refused_at_commit() { - let mut inner = SceneBuilder::new(); - inner.attach_labelled(placed(&part_at(0.0), 0.0), MASK_ALL, InstanceHitId::As(10)); - let inner = Arc::new(inner.commit()); - let mut b = SceneBuilder::new(); - b.attach_labelled( - placed(&inner, 0.0), - MASK_ALL, - InstanceHitId::Offset(u32::MAX - 5), - ); - let _ = b.commit(); - } - - /// The bound is exact enough to accept the largest offset that fits. - #[test] - fn the_largest_offset_that_fits_is_accepted() { - let mut inner = SceneBuilder::new(); - inner.attach_labelled(placed(&part_at(0.0), 0.0), MASK_ALL, InstanceHitId::As(10)); - let inner = Arc::new(inner.commit()); - let mut b = SceneBuilder::new(); - let base = u32::MAX - 11; - b.attach_labelled(placed(&inner, 0.0), MASK_ALL, InstanceHitId::Offset(base)); - assert_eq!(id_at(&b.commit(), 0.0), Some(base + 10)); - } - - #[test] - #[should_panic(expected = "only an instance can relabel")] - fn only_instances_take_labels() { - SceneBuilder::new().attach_labelled( - Geometry::Sphere { - center: Vec3A::ZERO, - radius: 1.0, - }, - MASK_ALL, - InstanceHitId::As(3), - ); - } - - /// The forwarding field sits in padding: an island-scale scene holds - /// tens of millions of `InstancePrim`s, so a byte here is gigabytes. - #[test] - fn instance_prim_is_not_grown_by_forwarding() { - assert_eq!(std::mem::size_of::(), 96); - } - - #[test] - fn instance_hits_report_instance_geom_id_and_inner_prim_id() { - let mut inner = SceneBuilder::new(); - inner.attach(Geometry::TriangleMesh { - vertices: vec![ - [-1.0, -1.0, 0.0], - [1.0, -1.0, 0.0], - [1.0, 1.0, 0.0], - [-1.0, 1.0, 0.0], - ], - indices: vec![[0, 1, 2], [0, 2, 3]], - normals: None, - }); - let inner = Arc::new(inner.commit()); - - let mut b = SceneBuilder::new(); - let _floor = b.attach(Geometry::Sphere { - center: Vec3A::new(0.0, -100.0, 0.0), - radius: 1.0, - }); - let inst = b.attach(Geometry::Instance { - scene: inner, - transform: Affine3A::from_translation(glam::Vec3::new(0.0, 0.0, 5.0)), - transform_end: None, - }); - let scene = b.commit(); - // Upper-left region → second triangle of the instanced mesh. - let hit = scene - .intersect( - &Ray::new(Vec3A::new(-0.5, 0.5, 0.0), Vec3A::Z), - 0.001, - f32::INFINITY, - ) - .expect("hit"); - assert_eq!(hit.geom_id, inst); - assert_eq!(hit.prim_id, 1); - } -} diff --git a/crates/crust-rt/src/scene/mod.rs b/crates/crust-rt/src/scene/mod.rs new file mode 100644 index 00000000..69b1dd85 --- /dev/null +++ b/crates/crust-rt/src/scene/mod.rs @@ -0,0 +1,908 @@ +//! The Embree-shaped public API: attach [`Geometry`] objects to a +//! [`SceneBuilder`], `commit()` into an immutable [`Scene`], query with +//! `intersect` / `occluded`. + +use crate::aabb::AABB; +use crate::bvh::{Bvh, Primitives}; +use crate::prim::{ + CubicCurvePrim, CurvePrim, CylinderPrim, DEGENERATE_VERTEX, DiskPrim, GeomTable, + InstanceMotion, InstancePrim, NO_ID_OFFSET, PrimHit, PrimNode, SpherePrim, TriangleRecord, + transformed_aabb, +}; +use crate::ray::{MASK_ALL, Ray, RayMask}; +use glam::{Affine3A, Vec3A}; +use std::sync::Arc; + +/// One round (sphere-swept) curve segment: a cone frustum tangent to the +/// spheres `(p0, r0)` and `(p1, r1)`, with spherical caps. +#[derive(Clone, Copy, Debug)] +pub struct CurveSegment { + pub p0: Vec3A, + pub p1: Vec3A, + pub r0: f32, + pub r1: f32, +} + +/// How a committed scene stores its triangle packets — see +/// [`SceneBuilder::commit_with`]. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] +pub enum PacketLayout { + /// Each packet carries its four triangles' vertices (192 bytes): the + /// fastest in cache. The layout before indexed packets existed. + Gathered, + /// Each packet carries vertex indices (92 bytes) and gathers from the + /// scene's shared vertex table at every test; bit-identical hits. A + /// memory trade, not a speed one: on the subdivision stress grid it is + /// 104 → 79 kernel bytes per triangle for 13% less render throughput, + /// and on a 4 M-triangle soup that does not fit the cache it is 22% + /// fewer kernel bytes for 30% slower traversal — the twelve dependent + /// vertex loads per packet cost more than the bandwidth they save. + Indexed, + /// The measured default: `Gathered`. The expectation was that a tree + /// too large for the cache would favour the smaller packet; measured + /// with `ray_throughput --layout` in and out of cache, no size does, so + /// the indexed layout is an explicit opt-in for a scene that otherwise + /// does not fit. + #[default] + Auto, +} + +/// What [`SceneBuilder::commit_with`] lets a caller choose about the build. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct CommitOptions { + /// The triangle packet layout. + pub layout: PacketLayout, +} + +impl Default for CommitOptions { + fn default() -> Self { + CommitOptions { + layout: PacketLayout::Auto, + } + } +} + +/// Exact bytes a committed [`Scene`] holds, by structure — the kernel's +/// side of a memory report. Counts `capacity`, not `len`, because unused +/// capacity is resident too, and deduplicates shared instanced scenes so +/// a prototype placed a thousand times is counted once. +/// +/// Only the kernel's own allocations: the application's material tables, +/// the USD stage and anything else outside `crust-rt` are not visible +/// from here. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] +pub struct MemoryFootprint { + /// The `PrimNode` arrays: spheres, disks, cylinders and linear curve + /// segments. + pub prim_nodes: usize, + /// Instances, 96 bytes each inline, plus the endpoint transforms of the + /// moving ones. + pub instances: usize, + /// Cubic curve spans, 96 bytes each inline. + pub cubic_spans: usize, + /// 4-wide BVH nodes. + pub bvh_nodes: usize, + pub leaves: usize, + /// Triangle SIMD packets of the gathered layout (the vertices, SoA). + pub packets: usize, + /// Triangle SIMD packets of the indexed layout (vertex indices). + pub packets_indexed: usize, + /// Leaf primitive indices. + pub indices: usize, + /// Triangle records: vertex indices, ids and mask, 24 bytes each. + pub triangle_records: usize, + /// The shared vertex table, 12 bytes per vertex. + pub vertices: usize, + /// Per-vertex shading normals of the meshes that carry them, 12 bytes + /// per vertex. + pub vertex_normals: usize, + /// The per-geometry table (where each geometry's entries start). + pub geometry_tables: usize, + /// Packet lanes in total — a count, not bytes, so `total` ignores it. + pub lanes: usize, + /// Packet lanes holding a triangle; `lanes_filled / lanes` is the fill. + pub lanes_filled: usize, +} + +impl MemoryFootprint { + pub fn total(&self) -> usize { + self.prim_nodes + + self.instances + + self.cubic_spans + + self.bvh_nodes + + self.leaves + + self.packets + + self.packets_indexed + + self.indices + + self.triangle_records + + self.vertices + + self.vertex_normals + + self.geometry_tables + } +} + +/// Top-level primitives of a committed [`Scene`], split by kind. See +/// [`Scene::primitive_breakdown`]. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] +pub struct PrimitiveBreakdown { + pub triangles: usize, + pub spheres: usize, + pub disks: usize, + pub cylinders: usize, + pub curve_segments: usize, + pub cubic_curve_spans: usize, + pub instances: usize, +} + +/// One authored cubic curve span — its own Bézier control points and end +/// radii, intersected analytically (`crate::curve::cubic_curve_intersect`) +/// rather than pre-flattened into several [`CurveSegment`]s. Round, same +/// as `RoundCurves` — this only changes how a span is stored and +/// intersected, not the surface it represents. +#[derive(Clone, Copy, Debug)] +pub struct CubicCurveSegment { + pub cp: [Vec3A; 4], + pub r0: f32, + pub r1: f32, +} + +/// A geometry to attach to a scene. The variants mirror Embree's geometry +/// types (the subset crust needs): triangle meshes, analytic spheres, +/// disks and open cylinders, round curves, and instances of another +/// committed scene. Instances nest: +/// an instanced scene may itself contain instances, and transforms, +/// normals and ray masks compose correctly through every level. +pub enum Geometry { + /// Unpadded `[f32; 3]` arrays, which is how the committed scene stores + /// them: a `Vec3A` is 16 bytes for 12 of data, and these arrays are + /// the bulk of a scene's memory. + TriangleMesh { + vertices: Vec<[f32; 3]>, + indices: Vec<[u32; 3]>, + /// Optional per-vertex shading normals; hits interpolate them by + /// the barycentrics (`SmoothTriangle` semantics). + normals: Option>, + }, + Sphere { + center: Vec3A, + radius: f32, + }, + /// A flat circular disk. `normal` names its front: hits report it as + /// the outward normal, so `RayHit::front_face` tells the two sides apart. + /// Need not be unit length; it is normalised at commit. + Disk { + center: Vec3A, + normal: Vec3A, + radius: f32, + }, + /// The side wall of a circular cylinder from `p0` to `p1` — open, with no + /// end caps. Hits report the radial outward normal. + Cylinder { + p0: Vec3A, + p1: Vec3A, + radius: f32, + }, + RoundCurves { + segments: Vec, + }, + /// Cubic curve spans, intersected as true curves instead of being + /// flattened to `RoundCurves` polylines — see [`CubicCurveSegment`]. + CubicCurves { + segments: Vec, + }, + /// Another committed scene placed by `transform` (local-to-world). + /// With `transform_end`, the placement interpolates linearly (per + /// matrix element) at the ray's shutter time — transform motion blur. + /// `transform` must be invertible. + Instance { + scene: Arc, + transform: Affine3A, + /// Boxed because it's `None` for the overwhelming majority of + /// instances (only motion-blurred placements set it): inline it + /// and every `Geometry` value — the enum is sized by its largest + /// variant — pays an extra 64 bytes it never uses. A scene with + /// millions of `PointInstancer` placements makes that the + /// dominant cost of the whole geometry table. + transform_end: Option>, + }, +} + +/// An intersection: distance, ray-facing normal (flipped to oppose the +/// ray, with `front_face` recording the original orientation), the hit +/// barycentrics where meaningful, and the IDs that let the application +/// map the hit back to its own data (materials, lights, …). +/// +/// For hits inside an [`Geometry::Instance`], `geom_id` is the +/// *instance's* id in the queried scene and `prim_id` the primitive index +/// within the instanced scene — the application maps per top-level +/// geometry. +/// +/// A curve hit also reports where along the curve it lies — `u` in `[0, 1]` +/// across its segment or cubic span — and `dpdu`, the curve's direction +/// there, unnormalised, in the queried scene's space (carried through every +/// instance by its local-to-world linear part, at the ray's time). Every +/// other hit's `dpdu` is zero. +#[derive(Clone, Copy, Debug)] +pub struct RayHit { + pub t: f32, + pub normal: Vec3A, + pub dpdu: Vec3A, + pub front_face: bool, + pub u: f32, + pub v: f32, + pub geom_id: u32, + pub prim_id: u32, +} + +/// Which `geom_id` a hit found inside an [`Geometry::Instance`] reports. +/// +/// By default an instance reports its *own* id and the inner ids are lost, +/// which is all a host that maps one material per top-level geometry needs. +/// A prototype of many parts wants more: to be placed as *one* instance +/// (so the BVH above it sees one box per placement, not one per part) and +/// still have a hit say which part it landed on. Embree answers with an +/// instance-id stack beside the inner `geomID`; this answers with one id, +/// computed as the hit passes back out through each instance level, which +/// keeps [`RayHit`] and the host's lookup a single index. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] +pub enum InstanceHitId { + /// The instance's own `geom_id` in the scene it is attached to. + #[default] + Own, + /// A fixed id, whatever the inner scene reported. + As(u32), + /// `base` plus the id the inner scene reported, so an inner scene whose + /// hits already carry `0..n` maps onto `base..base + n` here. Nests: + /// each level adds its own base. [`SceneBuilder::commit`] panics if + /// `base` plus the largest id the inner scene can report would leave + /// the id space, rather than let a hit wrap onto another geometry. + Offset(u32), +} + +/// Accumulates geometries, then builds the acceleration structure once in +/// [`SceneBuilder::commit`] (Embree's `rtcCommitScene`). +#[derive(Default)] +pub struct SceneBuilder { + geoms: Vec<(Geometry, RayMask, InstanceHitId)>, +} + +impl SceneBuilder { + pub fn new() -> Self { + Self::default() + } + + /// Attaches a geometry visible to every ray category; returns its + /// `geom_id` (dense, starting at 0 — usable as a table index). + pub fn attach(&mut self, geometry: Geometry) -> u32 { + self.attach_masked(geometry, MASK_ALL) + } + + /// Attaches a geometry visible only to ray categories in `mask`. + pub fn attach_masked(&mut self, geometry: Geometry, mask: RayMask) -> u32 { + self.attach_labelled(geometry, mask, InstanceHitId::Own) + } + + /// Attaches an instance whose hits report `label` rather than the + /// instance's own id (see [`InstanceHitId`]). Returns the instance's own + /// `geom_id` all the same: it still occupies a slot. + /// + /// # Panics + /// If `label` is not [`InstanceHitId::Own`] and `geometry` is not an + /// instance — only an instance has inner hits to relabel. + pub fn attach_labelled( + &mut self, + geometry: Geometry, + mask: RayMask, + label: InstanceHitId, + ) -> u32 { + assert!( + label == InstanceHitId::Own || matches!(geometry, Geometry::Instance { .. }), + "only an instance can relabel its hits" + ); + self.geoms.push((geometry, mask, label)); + (self.geoms.len() - 1) as u32 + } + + /// Reserves capacity for `additional` more geometries. Purely a + /// performance/memory hint — callers that know an upcoming batch size + /// (a `PointInstancer` with N placements, say) should use it: without + /// it, growing a multi-million-entry `Vec` by repeated doubling both + /// re-copies everything so far at each doubling and can leave up to + /// ~2x the final size over-allocated. + pub fn reserve(&mut self, additional: usize) { + self.geoms.reserve(additional); + } + + /// Number of geometries attached so far. + pub fn count(&self) -> usize { + self.geoms.len() + } + + /// Replaces the geometry already attached at `id`, keeping its mask. + /// + /// For callers that must claim a `geom_id` before they can decide what + /// geometry belongs in it. The importer needs this: whether a mesh is + /// better placed by an instance or baked into world-space triangles + /// depends on how many times it turns out to be placed, which is not + /// known until the whole stage has been walked — but `geom_id`s are + /// handed out in traversal order and are the key the host's material + /// table is indexed by, so they cannot be assigned later. + /// + /// Attach a placeholder, keep the id, and fill it in here once the + /// decision is made. Only valid before [`SceneBuilder::commit`], which is + /// enforced by taking `&mut self`. + /// + /// # Panics + /// If `id` was never attached. + pub fn set_geometry(&mut self, id: u32, geometry: Geometry) { + self.geoms[id as usize].0 = geometry; + } + + /// The ray mask the geometry at `id` was attached with. + /// + /// # Panics + /// If `id` was never attached. + pub fn mask(&self, id: u32) -> RayMask { + self.geoms[id as usize].1 + } + + /// Replaces the ray mask of the geometry already attached at `id`, + /// keeping the geometry. The same deferred-decision escape hatch as + /// [`SceneBuilder::set_geometry`], for a visibility that is only known + /// once the whole input has been read. Only valid before + /// [`SceneBuilder::commit`]. + /// + /// # Panics + /// If `id` was never attached. + pub fn set_mask(&mut self, id: u32, mask: RayMask) { + self.geoms[id as usize].1 = mask; + } + + /// A geometry that expands to no primitives — the placeholder to pair + /// with [`SceneBuilder::set_geometry`]. Allocates nothing. + pub fn empty_geometry() -> Geometry { + Geometry::TriangleMesh { + vertices: Vec::new(), + indices: Vec::new(), + normals: None, + } + } + + /// How many primitives a geometry expands into. An upper bound: the + /// expansion skips degenerate entries (out-of-range indices, an empty + /// instanced scene), so the real count can be lower. + fn prim_upper_bound(geom: &Geometry) -> usize { + match geom { + Geometry::TriangleMesh { indices, .. } => indices.len(), + Geometry::RoundCurves { segments } => segments.len(), + Geometry::CubicCurves { segments } => segments.len(), + Geometry::Sphere { .. } + | Geometry::Disk { .. } + | Geometry::Cylinder { .. } + | Geometry::Instance { .. } => 1, + } + } + + /// Expands every geometry into primitives and builds the BVH with the + /// default [`CommitOptions`]. + #[must_use = "the committed scene is the only way to intersect it"] + pub fn commit(self) -> Scene { + self.commit_with(CommitOptions::default()) + } + + /// [`SceneBuilder::commit`] with the build choices made by the caller. + #[must_use = "the committed scene is the only way to intersect it"] + pub fn commit_with(self, options: CommitOptions) -> Scene { + let n_geoms = self.geoms.len() as u32; + let mut sink = Expansion { + input: sized_primitives(&self.geoms), + has_motion: false, + // Largest id a hit in this scene can report. Every geometry can + // report its own id; labels can report more (see `add`). + max_hit_id: n_geoms.saturating_sub(1), + }; + for (geom_id, (geom, mask, label)) in self.geoms.into_iter().enumerate() { + sink.add(geom_id as u32, geom, mask, label); + } + let layout = match options.layout { + PacketLayout::Gathered | PacketLayout::Auto => crate::bvh::Layout::Gathered, + PacketLayout::Indexed => crate::bvh::Layout::Indexed, + }; + Scene { + bvh: Bvh::new(sink.input, layout), + n_geoms, + has_motion: sink.has_motion, + max_hit_id: sink.max_hit_id, + } + } +} + +/// The primitive arrays a commit fills, each sized exactly once from the +/// total the geometries will expand into, instead of letting per-geometry +/// `reserve` calls grow them by doubling. With many geometries that doubling +/// leaves up to ~2x over-allocated — and since a primitive node is 128 bytes, +/// on a scene of a few hundred thousand baked triangles the waste was tens of +/// MiB of genuinely committed memory (`MemoryFootprint` counts capacity, not +/// length, which is why it showed up). The bound can only over-shoot by the +/// number of degenerate primitives skipped, normally zero. +fn sized_primitives(geoms: &[(Geometry, RayMask, InstanceHitId)]) -> Primitives { + let total: usize = geoms + .iter() + .map(|(g, _, _)| SceneBuilder::prim_upper_bound(g)) + .sum(); + let (mut n_tris, mut n_verts, mut n_inst, mut n_cubic) = (0usize, 0usize, 0usize, 0usize); + for (g, _, _) in geoms { + match g { + Geometry::TriangleMesh { + vertices, indices, .. + } => { + n_tris += indices.len(); + n_verts += vertices.len(); + } + Geometry::Instance { .. } => n_inst += 1, + Geometry::CubicCurves { segments } => n_cubic += segments.len(), + _ => {} + } + } + assert!( + n_verts < u32::MAX as usize, + "{n_verts} vertices in one scene: the shared vertex table is indexed by u32" + ); + // Each non-triangle kind has its own array; size each exactly. + let n_nodes = total - n_tris - n_inst - n_cubic; + Primitives { + tris: Vec::with_capacity(n_tris), + vertices: Vec::with_capacity(n_verts), + normals: Vec::new(), + geoms: Vec::with_capacity(geoms.len()), + prims: Vec::with_capacity(n_nodes), + instances: Vec::with_capacity(n_inst), + instance_bounds: Vec::with_capacity(n_inst), + cubics: Vec::with_capacity(n_cubic), + order: Vec::with_capacity(total - n_tris), + } +} + +/// A commit in progress: the primitives the geometries expand into, and what +/// the committed scene reports about them. +struct Expansion { + input: Primitives, + /// Whether anything placed moves over the shutter. + has_motion: bool, + max_hit_id: u32, +} + +impl Expansion { + /// Expands geometry `geom_id` into primitives, and records its table. + /// + /// Every geometry records one, a skipped one included (the defaults): the + /// tables are indexed by `geom_id`, so a missing entry would hand every + /// later triangle mesh its neighbour's vertex and normal bases, and the + /// last one an index past the end. + fn add(&mut self, geom_id: u32, geom: Geometry, mask: RayMask, label: InstanceHitId) { + let mut table = GeomTable { + vertex_base: self.input.vertices.len() as u32, + tri_base: self.input.tris.len() as u32, + ..GeomTable::default() + }; + self.expand(geom_id, geom, mask, label, &mut table); + self.input.geoms.push(table); + } + + /// The primitives of one geometry; returns early, adding none, for an + /// invalid disk or cylinder and for an instance of an empty scene. + fn expand( + &mut self, + geom_id: u32, + geom: Geometry, + mask: RayMask, + label: InstanceHitId, + table: &mut GeomTable, + ) { + let input = &mut self.input; + match geom { + Geometry::TriangleMesh { + vertices, + indices, + normals, + } => { + // Normals are per vertex, parallel to the vertex run, + // and only when the array covers every vertex — a + // short one is ignored whole, as the per-triangle + // check it replaces ignored each triangle it fell + // short of. + let n = vertices.len(); + let vertex_base = table.vertex_base; + input.vertices.extend(vertices); + if let Some(mut ns) = normals.filter(|ns| ns.len() >= n) { + table.normal_base = input.normals.len() as u32; + ns.truncate(n); + input.normals.extend(ns); + } + for (prim_id, [i0, i1, i2]) in indices.into_iter().enumerate() { + let in_range = (i0 as usize) < n && (i1 as usize) < n && (i2 as usize) < n; + let v = if in_range { + [vertex_base + i0, vertex_base + i1, vertex_base + i2] + } else { + [DEGENERATE_VERTEX; 3] + }; + input.tris.push(TriangleRecord { + v, + geom_id, + prim_id: prim_id as u32, + mask: if in_range { mask } else { RayMask::NONE }, + }); + } + } + Geometry::Sphere { center, radius } => { + input.push_node(PrimNode::Sphere(SpherePrim { + center, + radius, + geom_id, + mask, + })); + } + Geometry::Disk { + center, + normal, + radius, + } => { + // Degenerate or non-finite: no front, no extent, or a + // position that would poison the bounds. + if !center.is_finite() + || !normal.is_finite() + || normal.length_squared() == 0.0 + || !radius.is_finite() + || radius <= 0.0 + { + return; + } + input.push_node(PrimNode::Disk(DiskPrim { + center, + normal: normal.normalize(), + radius, + geom_id, + mask, + })); + } + Geometry::Cylinder { p0, p1, radius } => { + let length = (p1 - p0).length(); + if !p0.is_finite() + || !p1.is_finite() + || !length.is_finite() + || length <= 0.0 + || !radius.is_finite() + || radius <= 0.0 + { + return; + } + input.push_node(PrimNode::Cylinder(CylinderPrim { + p0, + axis: (p1 - p0) / length, + length, + radius, + geom_id, + mask, + })); + } + Geometry::RoundCurves { segments } => { + let joints = curve_joints(segments.iter().map(|s| (s.p0, s.p1, s.r0.min(s.r1)))); + for (prim_id, (s, joints)) in segments.into_iter().zip(joints).enumerate() { + input.push_node(PrimNode::Curve(CurvePrim { + p0: s.p0.to_array(), + p1: s.p1.to_array(), + r0: s.r0, + r1: s.r1, + geom_id, + prim_id: prim_id as u32, + mask, + joints, + })); + } + } + Geometry::CubicCurves { segments } => { + let joints = + curve_joints(segments.iter().map(|s| (s.cp[0], s.cp[3], s.r0.min(s.r1)))); + for (prim_id, (s, joints)) in segments.into_iter().zip(joints).enumerate() { + input.push_cubic(CubicCurvePrim { + cp: s.cp, + r0: s.r0, + r1: s.r1, + geom_id, + prim_id: prim_id as u32, + mask, + joints, + }); + } + } + Geometry::Instance { + scene, + transform, + transform_end, + } => { + let Some(inner_bounds) = scene.bounds() else { + return; // empty instanced scene + }; + let w2l = transform.inverse(); + let bounds = match &transform_end { + Some(end) => AABB::surrounding_box( + transformed_aabb(&inner_bounds, &transform), + transformed_aabb(&inner_bounds, end), + ), + None => transformed_aabb(&inner_bounds, &transform), + }; + // A scene moves if this placement is blurred, or if the + // thing being placed already moves. The inner scene + // carries its own committed flag, so this stays O(1) per + // instance however deeply they nest. + self.has_motion |= transform_end.is_some() || scene.has_motion(); + let (geom_id, id_offset) = match label { + InstanceHitId::Own => (geom_id, NO_ID_OFFSET), + InstanceHitId::As(id) => { + assert!(id != crate::INVALID_ID, "hit id {id} is reserved"); + self.max_hit_id = self.max_hit_id.max(id); + (id, NO_ID_OFFSET) + } + InstanceHitId::Offset(base) => { + // Checked here, once per instance, so the hit + // path can add without a branch: an offset that + // could carry an inner id past the id space + // would otherwise wrap onto an unrelated + // geometry, and a host would shade it with that + // geometry's material. + let top = base + .checked_add(scene.max_hit_id) + .filter(|&top| top != crate::INVALID_ID) + .unwrap_or_else(|| { + panic!( + "id offset {base} + inner ids up to {} overflows the \ + geom_id space", + scene.max_hit_id + ) + }); + self.max_hit_id = self.max_hit_id.max(top); + (geom_id, base) + } + }; + input.push_instance( + InstancePrim { + scene, + w2l, + motion: transform_end.map(|end| { + Box::new(InstanceMotion { + l2w: transform, + l2w_end: *end, + }) + }), + geom_id, + id_offset, + mask, + }, + bounds, + ); + } + } + } +} + +/// A committed, immutable scene. Queries are `&self` and thread-safe. +pub struct Scene { + bvh: Bvh, + n_geoms: u32, + has_motion: bool, + /// Largest `geom_id` a hit in this scene can report — a bound, computed + /// at commit, that lets an `Offset` label placing this scene be checked + /// once instead of on every hit. + max_hit_id: u32, +} + +impl Scene { + /// Closest hit in `(t_min, t_max)`, or `None` (Embree's + /// `rtcIntersect1`). + #[must_use] + pub fn intersect(&self, ray: &Ray, t_min: f32, t_max: f32) -> Option { + let hit = self.bvh.hit(ray, t_min, t_max)?; + let front_face = ray.dir.dot(hit.outward) < 0.0; + Some(RayHit { + t: hit.t, + normal: if front_face { + hit.outward + } else { + -hit.outward + }, + dpdu: hit.dpdu, + front_face, + u: hit.u, + v: hit.v, + geom_id: hit.geom_id, + prim_id: hit.prim_id, + }) + } + + /// Does the ray hit *anything* in `(t_min, t_max)`? Early-exit + /// traversal — the shadow-ray fast path (Embree's `rtcOccluded1`). + #[must_use] + pub fn occluded(&self, ray: &Ray, t_min: f32, t_max: f32) -> bool { + self.bvh.hit_any(ray, t_min, t_max) + } + + /// What the top-level instances `ids` are: `(geom_id, world bounds, + /// inner top-level primitive count, how many top-level instances share + /// the same inner scene)`. Diagnostic for a top level that will not + /// cull, paired with [`crate::traversal_stats::top_level_descents`]; + /// a linear scan, so ask once. + #[cfg(feature = "traversal-stats")] + pub fn describe_instances( + &self, + ids: &std::collections::HashSet, + ) -> Vec<(u32, AABB, usize, usize)> { + let mut sharing = std::collections::HashMap::<*const Scene, usize>::new(); + let mut found = Vec::new(); + for inst in self.bvh.instances() { + *sharing.entry(Arc::as_ptr(&inst.scene)).or_insert(0) += 1; + if ids.contains(&inst.geom_id) { + let bounds = inst + .approx_world_bounds() + .unwrap_or(AABB::new(Vec3A::ZERO, Vec3A::ZERO)); + found.push(( + inst.geom_id, + bounds, + Arc::as_ptr(&inst.scene), + inst.scene.primitive_count(), + )); + } + } + found + .into_iter() + .map(|(id, b, ptr, n)| (id, b, n, sharing[&ptr])) + .collect() + } + + /// World bounds of everything in the scene; `None` when empty. + pub fn bounds(&self) -> Option { + self.bvh.bounds() + } + + /// Number of attached geometries (`geom_id`s are `0..count`). + pub fn geometry_count(&self) -> u32 { + self.n_geoms + } + + /// Does anything in this scene move over the shutter interval — i.e. can + /// `ray.time` change what a query returns? + /// + /// Transform motion blur is the only thing that reads `ray.time` + /// (`InstancePrim::transforms_at`), so when this is `false` the shutter + /// coordinate is unobservable and a host need not sample it. That is + /// worth asking about: drawing one costs a full 4-dimensional + /// quasi-random sample, which was 4.2% of the render on a static scene. + /// + /// True if any instance carries an end-of-shutter transform, at any depth + /// of nesting (each level folds in its inner scene's answer at commit). + pub fn has_motion(&self) -> bool { + self.has_motion + } + + /// Number of primitives the geometries expanded into. + pub fn primitive_count(&self) -> usize { + self.bvh.prim_count() + } + + /// Top-level primitives split by kind, for reporting. Instances count + /// as one primitive each and are *not* descended into — the instanced + /// scene's own contents are its own `Scene`'s business, and a + /// prototype shared by a thousand placements would otherwise be + /// counted a thousand times. + pub fn primitive_breakdown(&self) -> PrimitiveBreakdown { + self.bvh.primitive_breakdown() + } + + /// Primitives actually resident in memory: like + /// [`Scene::primitive_breakdown`], but descending into instanced + /// scenes, counting each distinct prototype **once** however many + /// placements reference it. + /// + /// This is the count that tracks memory. `primitive_breakdown` says + /// what the top-level BVH traverses; this says what is stored. For an + /// instance-heavy scene the two differ enormously, and the gap is the + /// whole benefit of instancing. + pub fn unique_primitive_breakdown(&self) -> PrimitiveBreakdown { + let mut visited = std::collections::HashSet::new(); + let mut acc = PrimitiveBreakdown::default(); + self.accumulate_unique_into(&mut visited, &mut acc); + acc + } + + pub(crate) fn accumulate_unique_into( + &self, + visited: &mut std::collections::HashSet, + acc: &mut PrimitiveBreakdown, + ) { + self.bvh.accumulate_unique(visited, acc); + } + + /// How big this scene's top-level primitives are relative to the + /// scene itself: `(count, scene diagonal, mean prim diagonal, max prim + /// diagonal)`. + /// + /// Diagnostic for a BVH that will not cull. A hierarchy can only + /// separate primitives whose bounds are small against the whole; when + /// the mean ratio approaches 1 every box covers everything, no split + /// can divide them, and traversal degenerates to a linear scan however + /// good the builder is. + pub fn primitive_extents(&self) -> (usize, f32, f32, f32) { + let scene_diag = self + .bvh + .bounds() + .map(|b| (b.maximum - b.minimum).length()) + .unwrap_or(0.0); + let (n, sum, max) = self.bvh.primitive_extent_sum(); + let mean = if n == 0 { 0.0 } else { sum / n as f32 }; + (n, scene_diag, mean, max) + } + + /// The three vertices of triangle `prim_id` of the triangle mesh + /// `geom_id`, in this scene's own space — local space for a scene that + /// is placed through instances. `None` when `geom_id` is not a + /// triangle mesh of this scene, `prim_id` is past its triangles, or + /// the triangle's attached indices were out of range. + /// + /// What lets an application derive per-hit quantities (a tangent + /// frame, a texture density) from the geometry it attached instead of + /// storing them per triangle. + pub fn triangle_vertices(&self, geom_id: u32, prim_id: u32) -> Option<[Vec3A; 3]> { + self.bvh.triangle_vertices(geom_id, prim_id) + } + + /// Exact resident bytes of this scene and every distinct scene it + /// instances — see [`MemoryFootprint`]. + pub fn memory_footprint(&self) -> MemoryFootprint { + let mut visited = std::collections::HashSet::new(); + let mut acc = MemoryFootprint::default(); + self.accumulate_footprint_into(&mut visited, &mut acc); + acc + } + + pub(crate) fn accumulate_footprint_into( + &self, + visited: &mut std::collections::HashSet, + acc: &mut MemoryFootprint, + ) { + self.bvh.accumulate_footprint(visited, acc); + } + + /// Internal closest-hit that keeps the *outward* (unoriented) normal, + /// so instance transforms can map it without re-deriving orientation. + pub(crate) fn intersect_outward(&self, ray: &Ray, t_min: f32, t_max: f32) -> Option { + self.bvh.hit(ray, t_min, t_max) + } +} + +/// Which ends of each segment of a curve batch continue into its neighbour +/// in the batch: segment `i`'s end and segment `i + 1`'s start, where they +/// meet. A strand is attached as consecutive segments (or spans), and a cap +/// where two of them meet is buried in the next one's body except on the +/// outside of a bend, so a ray passing out of tubes must not meet it. Where +/// they meet is decided within float rounding of the coordinates (a B-spline +/// converted to Bézier form shares its span ends only to the last ulps) and +/// a thousandth of the radius: two strands whose ends merely touch are not +/// one strand to any visible degree. +fn curve_joints(ends: impl Iterator) -> Vec { + let ends: Vec<_> = ends.collect(); + let meet = |a: Vec3A, b: Vec3A, r: f32| { + let scale = a.abs().max(b.abs()).max_element(); + (a - b).abs().max_element() <= 4.0 * f32::EPSILON * scale + 1e-3 * r + }; + let mut joints = vec![0u8; ends.len()]; + for i in 1..ends.len() { + let ((_, prev_end, r0), (start, _, r1)) = (ends[i - 1], ends[i]); + if meet(prev_end, start, r0.min(r1)) { + joints[i - 1] |= crate::curve::JOINED_END; + joints[i] |= crate::curve::JOINED_START; + } + } + joints +} + +#[cfg(test)] +mod tests; diff --git a/crates/crust-rt/src/scene/tests.rs b/crates/crust-rt/src/scene/tests.rs new file mode 100644 index 00000000..f7e04194 --- /dev/null +++ b/crates/crust-rt/src/scene/tests.rs @@ -0,0 +1,991 @@ +use super::*; +use crate::ray::{MASK_CAMERA, MASK_SHADOW}; +use glam::Mat4; + +fn unit_sphere_scene() -> Arc { + let mut b = SceneBuilder::new(); + b.attach(Geometry::Sphere { + center: Vec3A::ZERO, + radius: 1.0, + }); + Arc::new(b.commit()) +} + +/// A disk is hit inside its radius and nowhere else, from either side, +/// and `front_face` names the side its `normal` points to. +#[test] +fn disk_is_flat_round_and_knows_its_front() { + let mut b = SceneBuilder::new(); + let id = b.attach(Geometry::Disk { + center: Vec3A::new(0.0, 0.0, 2.0), + normal: Vec3A::new(0.0, 0.0, -3.0), // unnormalised on purpose + radius: 1.0, + }); + let s = b.commit(); + + let from_front = s + .intersect( + &Ray::new(Vec3A::new(0.5, 0.5, 0.0), Vec3A::Z), + 0.0, + f32::INFINITY, + ) + .expect("inside the radius"); + assert_eq!(from_front.geom_id, id); + assert!((from_front.t - 2.0).abs() < 1e-6); + assert!( + from_front.front_face, + "the ray arrives on the -Z (front) side" + ); + assert!(from_front.normal.abs_diff_eq(-Vec3A::Z, 1e-6)); + + let from_back = s + .intersect( + &Ray::new(Vec3A::new(0.0, 0.0, 5.0), -Vec3A::Z), + 0.0, + f32::INFINITY, + ) + .expect("a disk is visible from behind too"); + assert!(!from_back.front_face); + + // Just outside the radius (0.72² + 0.72² > 1) misses; parallel misses. + assert!( + s.intersect( + &Ray::new(Vec3A::new(0.72, 0.72, 0.0), Vec3A::Z), + 0.0, + f32::INFINITY + ) + .is_none() + ); + assert!( + s.intersect( + &Ray::new(Vec3A::new(-3.0, 0.0, 2.0), Vec3A::X), + 0.0, + f32::INFINITY + ) + .is_none() + ); + + // Bounds are exact in the plane and padded across it. + let bb = s.bounds().unwrap(); + assert!((bb.maximum.x - 1.0).abs() < 1e-6 && (bb.minimum.y + 1.0).abs() < 1e-6); + assert!(bb.maximum.z > bb.minimum.z); +} + +/// Degenerate or non-finite disks and cylinders are skipped at commit +/// rather than handed to the BVH, where one infinite bound would poison +/// every box above it. +#[test] +fn invalid_disks_and_cylinders_are_skipped() { + let mut b = SceneBuilder::new(); + for g in [ + Geometry::Disk { + center: Vec3A::ZERO, + normal: Vec3A::ZERO, + radius: 1.0, + }, + Geometry::Disk { + center: Vec3A::ZERO, + normal: Vec3A::Z, + radius: -1.0, + }, + Geometry::Disk { + center: Vec3A::ZERO, + normal: Vec3A::Z, + radius: f32::INFINITY, + }, + Geometry::Disk { + center: Vec3A::splat(f32::NAN), + normal: Vec3A::Z, + radius: 1.0, + }, + Geometry::Cylinder { + p0: Vec3A::ZERO, + p1: Vec3A::ZERO, + radius: 1.0, + }, + Geometry::Cylinder { + p0: Vec3A::ZERO, + p1: Vec3A::X, + radius: 0.0, + }, + Geometry::Cylinder { + p0: Vec3A::ZERO, + p1: Vec3A::splat(f32::INFINITY), + radius: 1.0, + }, + ] { + b.attach(g); + } + let s = b.commit(); + assert_eq!( + s.primitive_breakdown().disks + s.primitive_breakdown().cylinders, + 0 + ); + assert!(s.bounds().is_none()); +} + +/// An open tube: the wall is hit from outside with an outward normal, +/// from inside as a back face, and the ends are open. +#[test] +fn cylinder_is_an_open_tube() { + let mut b = SceneBuilder::new(); + b.attach(Geometry::Cylinder { + p0: Vec3A::new(-1.0, 0.0, 0.0), + p1: Vec3A::new(1.0, 0.0, 0.0), + radius: 0.5, + }); + let s = b.commit(); + + let outside = s + .intersect( + &Ray::new(Vec3A::new(0.3, 0.0, -4.0), Vec3A::Z), + 0.0, + f32::INFINITY, + ) + .expect("the wall"); + assert!((outside.t - 3.5).abs() < 1e-5); + assert!(outside.front_face); + assert!(outside.normal.abs_diff_eq(-Vec3A::Z, 1e-5)); + + let inside = s + .intersect( + &Ray::new(Vec3A::new(0.3, 0.0, 0.0), Vec3A::Y), + 0.0, + f32::INFINITY, + ) + .expect("the wall, from within"); + assert!((inside.t - 0.5).abs() < 1e-5); + assert!(!inside.front_face, "the inside of a tube is its back face"); + + // Down the axis there are no caps to hit; past the end, no wall. + assert!( + s.intersect( + &Ray::new(Vec3A::new(-5.0, 0.1, 0.0), Vec3A::X), + 0.0, + f32::INFINITY + ) + .is_none() + ); + assert!( + s.intersect( + &Ray::new(Vec3A::new(1.2, 0.0, -4.0), Vec3A::Z), + 0.0, + f32::INFINITY + ) + .is_none() + ); + // An oblique ray entering past the end exits through the wall. + let oblique = s + .intersect( + &Ray::new(Vec3A::new(1.5, 0.0, 0.0), Vec3A::new(-1.0, 0.0, 1.0)), + 0.0, + f32::INFINITY, + ) + .expect("exits through the wall"); + assert!(!oblique.front_face); + + let bb = s.bounds().unwrap(); + assert!(bb.minimum.abs_diff_eq(Vec3A::new(-1.0, -0.5, -0.5), 1e-6)); + assert!(bb.maximum.abs_diff_eq(Vec3A::new(1.0, 0.5, 0.5), 1e-6)); + assert_eq!(s.primitive_breakdown().cylinders, 1); +} + +/// Placed through an instance, a disk under a non-uniform scale is an +/// ellipse — the path the importer takes for a squashed `DiskLight`. +#[test] +fn instanced_disk_becomes_an_ellipse() { + let mut inner = SceneBuilder::new(); + inner.attach(Geometry::Disk { + center: Vec3A::ZERO, + normal: -Vec3A::Z, + radius: 1.0, + }); + let inner = Arc::new(inner.commit()); + let mut b = SceneBuilder::new(); + b.attach(Geometry::Instance { + scene: inner, + transform: Affine3A::from_scale(glam::Vec3::new(3.0, 1.0, 1.0)), + transform_end: None, + }); + let s = b.commit(); + let hit = |x: f32| s.intersect(&Ray::new(Vec3A::new(x, 0.0, -1.0), Vec3A::Z), 0.0, 10.0); + assert!(hit(2.9).is_some_and(|h| h.front_face)); + assert!(hit(3.1).is_none()); +} + +/// A reserved slot must keep its `geom_id` (so ids stay dense and every +/// later attach is unperturbed) and contribute nothing until filled. +#[test] +fn reserved_slots_keep_ids_dense_and_stay_invisible() { + let mut b = SceneBuilder::new(); + let a = b.attach(Geometry::Sphere { + center: Vec3A::new(-5.0, 0.0, 0.0), + radius: 1.0, + }); + let placeholder = b.attach(SceneBuilder::empty_geometry()); + let c = b.attach(Geometry::Sphere { + center: Vec3A::new(5.0, 0.0, 0.0), + radius: 1.0, + }); + assert_eq!((a, placeholder, c), (0, 1, 2), "ids stay dense"); + + // Never filled in: the two spheres are all there is. + let scene = b.commit(); + assert_eq!(scene.geometry_count(), 3); + assert_eq!(scene.primitive_count(), 2); + + // Filling it in later puts real geometry under the id it claimed. + let mut b = SceneBuilder::new(); + b.attach(Geometry::Sphere { + center: Vec3A::new(-5.0, 0.0, 0.0), + radius: 1.0, + }); + let slot = b.attach(SceneBuilder::empty_geometry()); + b.set_geometry( + slot, + Geometry::Sphere { + center: Vec3A::ZERO, + radius: 1.0, + }, + ); + let scene = b.commit(); + assert_eq!(scene.primitive_count(), 2); + let hit = scene + .intersect( + &Ray::new(Vec3A::new(0.0, 0.0, -8.0), Vec3A::Z), + 1e-4, + f32::MAX, + ) + .expect("the filled-in sphere is hit"); + assert_eq!(hit.geom_id, slot, "and reports the id it reserved"); +} + +#[test] +fn ids_map_back_to_geometries() { + let mut b = SceneBuilder::new(); + let ball = b.attach(Geometry::Sphere { + center: Vec3A::new(-3.0, 0.0, 0.0), + radius: 1.0, + }); + let quad = b.attach(Geometry::TriangleMesh { + vertices: vec![ + [2.0, -1.0, -1.0], + [2.0, -1.0, 1.0], + [2.0, 1.0, 1.0], + [2.0, 1.0, -1.0], + ], + indices: vec![[0, 1, 2], [0, 2, 3]], + normals: None, + }); + let scene = b.commit(); + assert_eq!(scene.geometry_count(), 2); + assert_eq!(scene.primitive_count(), 3); + + let hit_ball = scene + .intersect( + &Ray::new(Vec3A::new(-3.0, 0.0, -5.0), Vec3A::Z), + 0.001, + f32::INFINITY, + ) + .expect("ball hit"); + assert_eq!(hit_ball.geom_id, ball); + + // Aim at the second triangle of the quad (upper-left half). + let hit_quad = scene + .intersect( + &Ray::new(Vec3A::new(0.0, 0.5, -0.5), Vec3A::X), + 0.001, + f32::INFINITY, + ) + .expect("quad hit"); + assert_eq!(hit_quad.geom_id, quad); + assert_eq!(hit_quad.prim_id, 1); +} + +#[test] +fn front_face_semantics_match_ray_side() { + let scene = unit_sphere_scene(); + let outside = scene + .intersect( + &Ray::new(Vec3A::new(0.0, 0.0, -5.0), Vec3A::Z), + 0.001, + 100.0, + ) + .expect("outside hit"); + assert!(outside.front_face); + assert!(outside.normal.abs_diff_eq(-Vec3A::Z, 1e-4)); + + // From inside the sphere the normal flips toward the origin. + let inside = scene + .intersect(&Ray::new(Vec3A::ZERO, Vec3A::Z), 0.001, 100.0) + .expect("inside hit"); + assert!(!inside.front_face); + assert!(inside.normal.abs_diff_eq(-Vec3A::Z, 1e-4)); +} + +/// Does the kernel already nest instances? An `Instance` holds an +/// `Arc`, and nothing stops that scene from containing +/// instances of its own — this asks whether the recursion actually +/// works end to end, or only looks like it should. +#[test] +fn motion_flag_reports_a_static_scene_as_static() { + let mut b = SceneBuilder::new(); + b.attach(Geometry::Sphere { + center: Vec3A::ZERO, + radius: 1.0, + }); + assert!(!b.commit().has_motion()); + + // A *static* placement of static geometry is still static. + let mut b = SceneBuilder::new(); + b.attach(Geometry::Instance { + scene: unit_sphere_scene(), + transform: Affine3A::from_translation(glam::Vec3::new(3.0, 0.0, 0.0)), + transform_end: None, + }); + assert!(!b.commit().has_motion()); +} + +/// The way `has_motion` can be wrong that matters: motion authored on an +/// *inner* instance, placed by an outer one that does not itself move. +/// If the flag only looked at its own `transform_end`, the renderer would +/// stop sampling the shutter and silently drop the blur. +#[test] +fn motion_flag_propagates_through_nesting() { + let leaf = unit_sphere_scene(); + + // Level 1: the sphere streaks along X over the shutter. + let mut mid = SceneBuilder::new(); + mid.attach(Geometry::Instance { + scene: leaf, + transform: Affine3A::IDENTITY, + transform_end: Some(Box::new(Affine3A::from_translation(glam::Vec3::new( + 4.0, 0.0, 0.0, + )))), + }); + let mid = Arc::new(mid.commit()); + assert!(mid.has_motion(), "the level that authored the motion"); + + // Level 2: a static placement of that moving scene. Still moving. + let mut root = SceneBuilder::new(); + root.attach(Geometry::Instance { + scene: Arc::clone(&mid), + transform: Affine3A::from_translation(glam::Vec3::new(0.0, 7.0, 0.0)), + transform_end: None, + }); + assert!( + root.commit().has_motion(), + "a static placement of moving geometry still moves" + ); +} + +#[test] +fn instances_nest() { + // Level 0: a unit sphere at the origin. + let leaf = unit_sphere_scene(); + + // Level 1: two spheres, at local x = ±2. + let mut mid = SceneBuilder::new(); + for x in [-2.0f32, 2.0] { + mid.attach(Geometry::Instance { + scene: Arc::clone(&leaf), + transform: Affine3A::from_translation(glam::Vec3::new(x, 0.0, 0.0)), + transform_end: None, + }); + } + let mid = Arc::new(mid.commit()); + + // Level 2: two copies of that pair, at world y = ±5. Four spheres + // in total, from one copy of the sphere's geometry. + let mut root = SceneBuilder::new(); + for y in [-5.0f32, 5.0] { + root.attach(Geometry::Instance { + scene: Arc::clone(&mid), + transform: Affine3A::from_translation(glam::Vec3::new(0.0, y, 0.0)), + transform_end: None, + }); + } + let scene = root.commit(); + + // Two top-level primitives hold four spheres. + assert_eq!(scene.primitive_count(), 2); + + for (x, y) in [(-2.0f32, -5.0f32), (2.0, -5.0), (-2.0, 5.0), (2.0, 5.0)] { + let ray = Ray::new(Vec3A::new(x, y, -8.0), Vec3A::Z); + let hit = scene + .intersect(&ray, 0.001, 100.0) + .unwrap_or_else(|| panic!("nested sphere at ({x}, {y}) was missed")); + assert!( + (hit.t - 7.0).abs() < 1e-3, + "nested sphere at ({x}, {y}): t = {} (want 7)", + hit.t + ); + assert!( + hit.normal.abs_diff_eq(-Vec3A::Z, 1e-4), + "nested normal wrong at ({x}, {y}): {:?}", + hit.normal + ); + assert!(scene.occluded(&ray, 0.001, 100.0)); + } + + // And nothing where the spheres are not. + assert!( + scene + .intersect( + &Ray::new(Vec3A::new(0.0, 0.0, -8.0), Vec3A::Z), + 0.001, + 100.0 + ) + .is_none(), + "hit between the nested spheres" + ); +} + +/// The reporting counts: the top-level view sees instances, the unique +/// view descends but counts a shared prototype only once — the whole +/// point being that four placed spheres cost one sphere of memory. +#[test] +fn unique_breakdown_counts_shared_prototypes_once() { + let leaf = unit_sphere_scene(); + let mut mid = SceneBuilder::new(); + for x in [-2.0f32, 2.0] { + mid.attach(Geometry::Instance { + scene: Arc::clone(&leaf), + transform: Affine3A::from_translation(glam::Vec3::new(x, 0.0, 0.0)), + transform_end: None, + }); + } + let mid = Arc::new(mid.commit()); + let mut root = SceneBuilder::new(); + for y in [-5.0f32, 5.0] { + root.attach(Geometry::Instance { + scene: Arc::clone(&mid), + transform: Affine3A::from_translation(glam::Vec3::new(0.0, y, 0.0)), + transform_end: None, + }); + } + let scene = root.commit(); + + // Top level: the two outer placements, no spheres visible yet. + let top = scene.primitive_breakdown(); + assert_eq!(top.instances, 2); + assert_eq!(top.spheres, 0); + + // Unique: descends both levels, but `mid` is one Arc shared by two + // placements and `leaf` one Arc shared by two more — so exactly one + // sphere is resident, reached through 2 + 2 instance primitives. + let unique = scene.unique_primitive_breakdown(); + assert_eq!(unique.spheres, 1, "shared prototype counted more than once"); + assert_eq!(unique.instances, 4); +} + +/// Nested instances must compose transforms in the right order, and +/// map normals back through both levels. A rotation at the outer level +/// and a non-uniform scale at the inner level do not commute, so this +/// fails loudly if the composition is inverted. +#[test] +fn nested_instances_compose_transforms_and_normals() { + // Inner: a unit sphere squashed to an ellipsoid by the mid level. + let leaf = unit_sphere_scene(); + let mut mid = SceneBuilder::new(); + mid.attach(Geometry::Instance { + scene: leaf, + // 2x along local X only. + transform: Affine3A::from_scale(glam::Vec3::new(2.0, 1.0, 1.0)), + transform_end: None, + }); + let mid = Arc::new(mid.commit()); + + // Outer: rotate that ellipsoid 90 degrees about Z, so its long + // axis ends up along world Y. + let mut root = SceneBuilder::new(); + root.attach(Geometry::Instance { + scene: mid, + transform: Affine3A::from_rotation_z(std::f32::consts::FRAC_PI_2), + transform_end: None, + }); + let scene = root.commit(); + + // Long axis is now Y: a ray down the Y axis meets the surface at + // |y| = 2, while one down X meets it at |x| = 1. + let along_y = scene + .intersect( + &Ray::new(Vec3A::new(0.0, -8.0, 0.0), Vec3A::Y), + 0.001, + 100.0, + ) + .expect("ray along Y must hit the rotated ellipsoid"); + assert!( + (along_y.t - 6.0).abs() < 1e-3, + "long axis is not along Y: t = {} (want 6)", + along_y.t + ); + let along_x = scene + .intersect( + &Ray::new(Vec3A::new(-8.0, 0.0, 0.0), Vec3A::X), + 0.001, + 100.0, + ) + .expect("ray along X must hit the rotated ellipsoid"); + assert!( + (along_x.t - 7.0).abs() < 1e-3, + "short axis is not along X: t = {} (want 7)", + along_x.t + ); + + // The normal at the Y pole points back down -Y; an inverse + // transpose applied at only one level would tilt it. + assert!( + along_y.normal.abs_diff_eq(-Vec3A::Y, 1e-4), + "nested normal not mapped through both levels: {:?}", + along_y.normal + ); +} + +/// A ray mask must gate at every level of nesting: hiding the outer +/// instance hides everything beneath it. +#[test] +fn nested_instances_respect_masks_at_each_level() { + let leaf = unit_sphere_scene(); + let mut mid = SceneBuilder::new(); + mid.attach_masked( + Geometry::Instance { + scene: leaf, + transform: Affine3A::IDENTITY, + transform_end: None, + }, + MASK_SHADOW, + ); + let mid = Arc::new(mid.commit()); + + let mut root = SceneBuilder::new(); + root.attach_masked( + Geometry::Instance { + scene: mid, + transform: Affine3A::IDENTITY, + transform_end: None, + }, + MASK_SHADOW | MASK_CAMERA, + ); + let scene = root.commit(); + + let ray = Ray::new(Vec3A::new(0.0, 0.0, -5.0), Vec3A::Z); + // The inner level only admits shadow rays, so a camera ray is + // rejected there even though the outer level would allow it. + assert!( + scene + .intersect(&ray.with_mask(MASK_CAMERA), 0.001, 100.0) + .is_none() + ); + assert!( + scene + .intersect(&ray.with_mask(MASK_SHADOW), 0.001, 100.0) + .is_some() + ); +} + +#[test] +fn masks_filter_by_ray_category() { + let mut b = SceneBuilder::new(); + b.attach_masked( + Geometry::Sphere { + center: Vec3A::ZERO, + radius: 1.0, + }, + MASK_SHADOW, + ); + let scene = b.commit(); + let ray = Ray::new(Vec3A::new(0.0, 0.0, -5.0), Vec3A::Z); + assert!( + scene + .intersect(&ray.with_mask(MASK_CAMERA), 0.001, 100.0) + .is_none() + ); + assert!( + scene + .intersect(&ray.with_mask(MASK_SHADOW), 0.001, 100.0) + .is_some() + ); + assert!(!scene.occluded(&ray.with_mask(MASK_CAMERA), 0.001, 100.0)); + assert!(scene.occluded(&ray.with_mask(MASK_SHADOW), 0.001, 100.0)); +} + +/// A mask replaced after attaching is the one the committed scene +/// filters by, exactly as if it had been attached with it. +#[test] +fn set_mask_replaces_the_attached_mask() { + let mut b = SceneBuilder::new(); + let id = b.attach(Geometry::Sphere { + center: Vec3A::ZERO, + radius: 1.0, + }); + assert_eq!(b.mask(id), MASK_ALL); + b.set_mask(id, MASK_CAMERA); + assert_eq!(b.mask(id), MASK_CAMERA); + let scene = b.commit(); + let ray = Ray::new(Vec3A::new(0.0, 0.0, -5.0), Vec3A::Z); + assert!( + scene + .intersect(&ray.with_mask(MASK_CAMERA), 0.001, 100.0) + .is_some() + ); + assert!(!scene.occluded(&ray.with_mask(MASK_SHADOW), 0.001, 100.0)); +} + +/// A geometry the commit skips — an invalid disk or cylinder, an instance of +/// an empty scene — still holds its `geom_id`'s slot in the geometry tables, +/// which are indexed by id: a triangle mesh attached after one reads its own +/// vertices and normals, not its neighbour's (or past the end of the table). +#[test] +fn a_skipped_geometry_keeps_its_table_slot() { + let empty = Arc::new(SceneBuilder::new().commit()); + let mut b = SceneBuilder::new(); + b.attach(Geometry::Disk { + center: Vec3A::ZERO, + normal: Vec3A::Z, + radius: 0.0, + }); + b.attach(Geometry::Instance { + scene: empty, + transform: glam::Affine3A::IDENTITY, + transform_end: None, + }); + let mesh = b.attach(Geometry::TriangleMesh { + vertices: vec![[-1.0, -1.0, 2.0], [1.0, -1.0, 2.0], [0.0, 1.0, 2.0]], + indices: vec![[0, 1, 2]], + normals: Some(vec![Vec3A::new(0.5, 0.0, -1.0).normalize().to_array(); 3]), + }); + let scene = b.commit(); + let hit = scene + .intersect( + &Ray::new(Vec3A::new(0.0, -0.5, 0.0), Vec3A::Z), + 0.001, + 100.0, + ) + .expect("the mesh is hit"); + assert_eq!(hit.geom_id, mesh); + assert!( + hit.normal + .abs_diff_eq(Vec3A::new(0.5, 0.0, -1.0).normalize(), 1e-5), + "the mesh shaded with another geometry's normals: {:?}", + hit.normal + ); +} + +#[test] +fn smooth_normals_interpolate() { + // One triangle with vertex normals fanned outward; the hit normal + // at an interior point must be a blend, not the face normal. + let mut b = SceneBuilder::new(); + b.attach(Geometry::TriangleMesh { + vertices: vec![[-1.0, -1.0, 2.0], [1.0, -1.0, 2.0], [0.0, 1.0, 2.0]], + indices: vec![[0, 1, 2]], + normals: Some(vec![ + Vec3A::new(-0.5, 0.0, -1.0).normalize().to_array(), + Vec3A::new(0.5, 0.0, -1.0).normalize().to_array(), + Vec3A::new(0.0, 0.5, -1.0).normalize().to_array(), + ]), + }); + let scene = b.commit(); + // Straight at the v1 corner region: x > 0 → normal tilts +x. + let hit = scene + .intersect( + &Ray::new(Vec3A::new(0.6, -0.7, 0.0), Vec3A::Z), + 0.001, + 100.0, + ) + .expect("hit"); + assert!( + hit.normal.x > 0.1, + "normal not interpolated: {:?}", + hit.normal + ); + assert!(hit.normal.z < 0.0); +} + +#[test] +fn translated_instance_matches_baked() { + let mut b = SceneBuilder::new(); + b.attach(Geometry::Instance { + scene: unit_sphere_scene(), + transform: Affine3A::from_translation(glam::Vec3::new(3.0, 0.0, 0.0)), + transform_end: None, + }); + let scene = b.commit(); + let ray = Ray::new(Vec3A::new(3.0, 0.0, -5.0), Vec3A::Z); + let hit = scene.intersect(&ray, 0.001, f32::INFINITY).expect("hit"); + assert!((hit.t - 4.0).abs() < 1e-4); + assert!(hit.normal.abs_diff_eq(-Vec3A::Z, 1e-4)); + assert!(scene.occluded(&ray, 0.001, f32::INFINITY)); + assert!(!scene.occluded(&ray, 0.001, 3.9)); +} + +#[test] +fn nonuniform_scale_transforms_normals_correctly() { + // Sphere scaled 2x in X: probe an oblique point where the naive + // (non inverse-transpose) normal mapping would be wrong. + let mut b = SceneBuilder::new(); + b.attach(Geometry::Instance { + scene: unit_sphere_scene(), + transform: Affine3A::from_scale(glam::Vec3::new(2.0, 1.0, 1.0)), + transform_end: None, + }); + let scene = b.commit(); + // Hit the ellipsoid straight down above x=1 (local x=0.5). + let hit = scene + .intersect( + &Ray::new(Vec3A::new(1.0, 5.0, 0.0), -Vec3A::Y), + 0.001, + f32::INFINITY, + ) + .expect("hit"); + // Implicit ellipsoid (x/2)^2 + y^2 + z^2 = 1: gradient at + // (1, sqrt(3)/2, 0) is proportional to (0.5, sqrt(3), 0). + let expected = Vec3A::new(0.5, 3.0f32.sqrt(), 0.0).normalize(); + assert!( + hit.normal.abs_diff_eq(expected, 1e-3), + "normal {:?} != expected {:?}", + hit.normal, + expected + ); +} + +#[test] +fn rotated_instance_hits_where_baked_triangle_would() { + let mut inner = SceneBuilder::new(); + inner.attach(Geometry::TriangleMesh { + vertices: vec![[-1.0, -1.0, 0.0], [1.0, -1.0, 0.0], [0.0, 1.0, 0.0]], + indices: vec![[0, 1, 2]], + normals: None, + }); + let mut b = SceneBuilder::new(); + b.attach(Geometry::Instance { + scene: Arc::new(inner.commit()), + transform: Affine3A::from_mat4( + Mat4::from_rotation_y(std::f32::consts::FRAC_PI_2) + * Mat4::from_translation(glam::Vec3::new(0.0, 0.0, 2.0)), + ), + transform_end: None, + }); + let scene = b.commit(); + // Local (0,0,2) maps to world (2,0,0); triangle now faces +X. + let hit = scene + .intersect( + &Ray::new(Vec3A::new(5.0, 0.0, 0.0), -Vec3A::X), + 0.001, + f32::INFINITY, + ) + .expect("hit"); + assert!((hit.t - 3.0).abs() < 1e-4); + assert!(hit.normal.abs_diff_eq(Vec3A::X, 1e-4)); +} + +#[test] +fn motion_blur_interpolates_position() { + let mut b = SceneBuilder::new(); + b.attach(Geometry::Instance { + scene: unit_sphere_scene(), + transform: Affine3A::IDENTITY, + transform_end: Some(Box::new(Affine3A::from_translation(glam::Vec3::new( + 4.0, 0.0, 0.0, + )))), + }); + let scene = b.commit(); + // At time 0 the sphere is at the origin... + let r0 = Ray::new(Vec3A::new(0.0, 0.0, -5.0), Vec3A::Z).with_time(0.0); + assert!(scene.intersect(&r0, 0.001, f32::INFINITY).is_some()); + // ...at time 1 it has moved to x=4... + let r1 = Ray::new(Vec3A::new(4.0, 0.0, -5.0), Vec3A::Z).with_time(1.0); + assert!(scene.intersect(&r1, 0.001, f32::INFINITY).is_some()); + let r1_origin = Ray::new(Vec3A::new(0.0, 0.0, -5.0), Vec3A::Z).with_time(1.0); + assert!(scene.intersect(&r1_origin, 0.001, f32::INFINITY).is_none()); + // ...and at time 0.5 it is halfway. + let rh = Ray::new(Vec3A::new(2.0, 0.0, -5.0), Vec3A::Z).with_time(0.5); + let hit = scene + .intersect(&rh, 0.001, f32::INFINITY) + .expect("halfway hit"); + assert!((hit.t - 4.0).abs() < 1e-4); + // The shutter-union bounding box covers both endpoints. + let bb = scene.bounds().unwrap(); + assert!(bb.minimum.x <= -1.0 && bb.maximum.x >= 5.0); +} + +/// A unit sphere at `x` in a one-geometry scene: a stand-in "part". +fn part_at(x: f32) -> Arc { + let mut b = SceneBuilder::new(); + b.attach(Geometry::Sphere { + center: Vec3A::new(x, 0.0, 0.0), + radius: 0.5, + }); + Arc::new(b.commit()) +} + +fn placed(scene: &Arc, z: f32) -> Geometry { + Geometry::Instance { + scene: scene.clone(), + transform: Affine3A::from_translation(glam::Vec3::new(0.0, 0.0, z)), + transform_end: None, + } +} + +/// The id a ray down +Z at `x` reports, if it hits. +fn id_at(scene: &Scene, x: f32) -> Option { + scene + .intersect( + &Ray::new(Vec3A::new(x, 0.0, -10.0), Vec3A::Z), + 0.001, + f32::INFINITY, + ) + .map(|h| h.geom_id) +} + +/// The shape the importer builds for a many-part prototype: parts +/// labelled `As(k)` inside a group, the group placed several times under +/// `Offset(0)` inside a scatter, and the scatter attached at the top +/// level under `Offset(base)`. Every hit must come out as `base + k`, +/// whichever placement it landed in. +#[test] +fn labelled_instances_compose_offsets_through_nesting() { + let mut group = SceneBuilder::new(); + for k in 0..3u32 { + // Attached out of order so a part's own id is never its label. + group.attach_labelled( + placed(&part_at(k as f32 * 2.0), 0.0), + MASK_ALL, + InstanceHitId::As(2 - k), + ); + } + let group = Arc::new(group.commit()); + + let mut scatter = SceneBuilder::new(); + for z in [0.0, 5.0] { + scatter.attach_labelled(placed(&group, z), MASK_ALL, InstanceHitId::Offset(0)); + } + let scatter = Arc::new(scatter.commit()); + + let mut top = SceneBuilder::new(); + let other = top.attach(Geometry::Sphere { + center: Vec3A::new(-5.0, 0.0, 0.0), + radius: 0.5, + }); + let base = 10; + let own = top.attach_labelled(placed(&scatter, 0.0), MASK_ALL, InstanceHitId::Offset(base)); + let top = top.commit(); + + assert_eq!(own, 1, "a labelled instance still takes its own slot"); + assert_eq!(id_at(&top, -5.0), Some(other)); + assert_eq!(id_at(&top, 0.0), Some(base + 2)); + assert_eq!(id_at(&top, 2.0), Some(base + 1)); + assert_eq!(id_at(&top, 4.0), Some(base)); + assert_eq!(id_at(&top, 1.0), None); +} + +/// `As` overrides the inner id outright, and `Own` still reports the +/// instance's slot even when what it places forwards. +#[test] +fn as_and_own_labels_override_inner_ids() { + let mut inner = SceneBuilder::new(); + inner.attach_labelled( + placed(&part_at(0.0), 0.0), + MASK_ALL, + InstanceHitId::Offset(7), + ); + let inner = Arc::new(inner.commit()); + + let mut b = SceneBuilder::new(); + let own = b.attach(placed(&inner, 0.0)); + let scene = b.commit(); + assert_eq!(id_at(&scene, 0.0), Some(own)); + + let mut b = SceneBuilder::new(); + b.attach_labelled(placed(&inner, 0.0), MASK_ALL, InstanceHitId::As(42)); + assert_eq!(id_at(&b.commit(), 0.0), Some(42)); +} + +/// An offset that could carry an inner id past the id space is refused +/// at commit: on the hit path it would wrap onto an unrelated id, and a +/// host would shade the hit with that geometry's material. +#[test] +#[should_panic(expected = "overflows the geom_id space")] +fn an_offset_that_could_overflow_is_refused_at_commit() { + let mut inner = SceneBuilder::new(); + inner.attach_labelled(placed(&part_at(0.0), 0.0), MASK_ALL, InstanceHitId::As(10)); + let inner = Arc::new(inner.commit()); + let mut b = SceneBuilder::new(); + b.attach_labelled( + placed(&inner, 0.0), + MASK_ALL, + InstanceHitId::Offset(u32::MAX - 5), + ); + let _ = b.commit(); +} + +/// The bound is exact enough to accept the largest offset that fits. +#[test] +fn the_largest_offset_that_fits_is_accepted() { + let mut inner = SceneBuilder::new(); + inner.attach_labelled(placed(&part_at(0.0), 0.0), MASK_ALL, InstanceHitId::As(10)); + let inner = Arc::new(inner.commit()); + let mut b = SceneBuilder::new(); + let base = u32::MAX - 11; + b.attach_labelled(placed(&inner, 0.0), MASK_ALL, InstanceHitId::Offset(base)); + assert_eq!(id_at(&b.commit(), 0.0), Some(base + 10)); +} + +#[test] +#[should_panic(expected = "only an instance can relabel")] +fn only_instances_take_labels() { + SceneBuilder::new().attach_labelled( + Geometry::Sphere { + center: Vec3A::ZERO, + radius: 1.0, + }, + MASK_ALL, + InstanceHitId::As(3), + ); +} + +/// The forwarding field sits in padding: an island-scale scene holds +/// tens of millions of `InstancePrim`s, so a byte here is gigabytes. +#[test] +fn instance_prim_is_not_grown_by_forwarding() { + assert_eq!(std::mem::size_of::(), 96); +} + +#[test] +fn instance_hits_report_instance_geom_id_and_inner_prim_id() { + let mut inner = SceneBuilder::new(); + inner.attach(Geometry::TriangleMesh { + vertices: vec![ + [-1.0, -1.0, 0.0], + [1.0, -1.0, 0.0], + [1.0, 1.0, 0.0], + [-1.0, 1.0, 0.0], + ], + indices: vec![[0, 1, 2], [0, 2, 3]], + normals: None, + }); + let inner = Arc::new(inner.commit()); + + let mut b = SceneBuilder::new(); + let _floor = b.attach(Geometry::Sphere { + center: Vec3A::new(0.0, -100.0, 0.0), + radius: 1.0, + }); + let inst = b.attach(Geometry::Instance { + scene: inner, + transform: Affine3A::from_translation(glam::Vec3::new(0.0, 0.0, 5.0)), + transform_end: None, + }); + let scene = b.commit(); + // Upper-left region → second triangle of the instanced mesh. + let hit = scene + .intersect( + &Ray::new(Vec3A::new(-0.5, 0.5, 0.0), Vec3A::Z), + 0.001, + f32::INFINITY, + ) + .expect("hit"); + assert_eq!(hit.geom_id, inst); + assert_eq!(hit.prim_id, 1); +} diff --git a/docs/architecture.md b/docs/architecture.md index 5faf28be..9b69c037 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -86,7 +86,7 @@ crust-render::main │ backward gather: MIS-weighted radiance, guiding training samples │ AOV instantiation only: the first hit → the unit's AOV planes └─ write EXR (linear) + PNG (tone-mapped) — crust-render only - no products: write_rgb_file at -o; products: one scanline EXR each (main.rs, `mod products`) + no products: write_rgb_file at -o; products: one scanline EXR each (main.rs, products.rs) ``` Path guiding (`render_guided`) and adaptive sampling wrap the same per-pixel @@ -106,7 +106,7 @@ both sides must keep; the contract lives in the doc comment at the definition. | seam | defined in | implemented by | contract in one line | |------|-----------|----------------|----------------------| -| `crust_rt::Geometry`, `SceneBuilder`, `Scene` | `crust-rt/src/scene.rs` | the kernel | Embree-shaped: attach, `commit()`, `intersect` / `occluded`; hits are `(geom_id, prim_id)` only | +| `crust_rt::Geometry`, `SceneBuilder`, `Scene` | `crust-rt/src/scene/` | the kernel | Embree-shaped: attach, `commit()`, `intersect` / `occluded`; hits are `(geom_id, prim_id)` only | | `WorldBuilder` / `World` | `crust-core/src/rt_world.rs` | — | pairs each `geom_id` with its material and per-triangle side tables (Ptex faces, UVs, density) | | `AssetLoader` | `crust-core/src/scene.rs` | `crust_assets::FileAssets`, `NoAssets` | the host decodes; returning `None` means "fall back", never an error | | `Texture2D` (= `crust_mtlx::Texture`), `PtexTexture` | `crust-mtlx/src/texture.rs`, `crust-core/src/texture.rs` | `UvTexture`, `StreamingTexture`, `PtexColor`, `PtexStream` | linear values out; unwrapped UVs in (UDIM addressing is the host's) | @@ -120,7 +120,7 @@ both sides must keep; the contract lives in the doc comment at the definition. | area | modules | |------|---------| | scene description | `scene.rs` (`Scene`, `AssetLoader`, `UsdImportOptions`), `camera.rs`, `world.rs` (procedural fallback scene) | -| USD import | `scene/usd_import/` — module map in its `mod.rs`; `scene/subdiv.rs` (OpenSubdiv refinement); `scene/displace.rs` (scalar displacement of tessellated meshes, once per distinct mesh) | +| USD import | `scene/usd_import/` — module map in its `mod.rs`; `scene/subdiv/` (OpenSubdiv refinement: `uniform`, per-face `adaptive`, `topology`, `normals`); `scene/displace.rs` (scalar displacement of tessellated meshes, once per distinct mesh) | | geometry bridge | `rt_world.rs` (`World`, side tables), `hittable.rs` (`HitRecord`), `ray.rs` (`Ray`, `RayCone`, ray masks), `aabb.rs` (re-export of the kernel's) | | AOVs | `aov.rs` (the source vocabulary, `AovRequest`, the per-unit planes and the full-frame `AovFilm`); products resolved in `scene/usd_import/products.rs`; `lpe/` (OSL light path expressions: parser, one DFA per render); `tracer/route.rs` (routing a path's light into the expressions, and the albedo) | | integrator | `tracer/` — `mod.rs` (`Renderer`: passes, tiles, guiding schedule), `path.rs` (`trace_path`, NEE, MIS weights, QMC domain keys), `settings.rs` (`RenderSettings`, `SamplingStrategy`); `filter.rs` (pixel filter importance sampling), `buffer.rs` | @@ -323,9 +323,10 @@ Still open, roughly in order of payoff: the kernel's type. 2. **Test files over 1 500 lines** (`usd_scene.rs`, `usd_inline.rs`, `crust-mtlx/tests/graph.rs`) would split naturally by schema family, the - way the importer now does. The largest source files left are - `crust-rt/src/scene.rs` (1 350), `stats.rs` (1 360), `materialx.rs` - (1 430) and `usd_import/mesh.rs` (1 410); none is urgent. + way the importer now does. `subdiv`, `crust-mtlx`'s `eval` and `surface`, + and `crust-rt`'s `scene` are directories now, their tests in their own + files; the largest source files left are `usd_import/mesh.rs`, + `tracer/path.rs` and `stats.rs`; none is urgent. 3. **Hot-path splits need a callgrind, not an eye.** Any further move inside `tracer/path.rs` or `bvh/mod.rs` should repeat the per-function instruction comparison above: the integrator is monomorphised on diff --git a/docs/color_management.md b/docs/color_management.md index 84fc1ac6..8abb2121 100644 --- a/docs/color_management.md +++ b/docs/color_management.md @@ -274,7 +274,7 @@ takes the texture's `sourceColorSpace` instead (see the textures table). | `crust:openpbr` — all 8 colour fields[^1] | `usd_import/materials.rs`, `decode_crust_openpbr` | **none** unless `colorSpace` metadata names a space | ✅ intentional — native format is authored in the working space | | `crust:openpbr` `subsurfaceRadiusScale` | same | never — a per-channel radius multiplier, not a colour | ✅ | | MaterialX `uniform_edf.color` | `crust-mtlx/src/bsdf.rs`, `edf_walk` | whatever the feeding node declares | ✅ correct per MaterialX | -| MaterialX surface-node colours (`base_color`, `specular_color`, `coat_color`, …) and leaf colours, **literal** | `crust-mtlx/src/eval.rs`, at compile time through `Host::convert_color` | the effective `colorspace` → working; none when no scope declares one | ✅ correct per MaterialX | +| MaterialX surface-node colours (`base_color`, `specular_color`, `coat_color`, …) and leaf colours, **literal** | `crust-mtlx/src/eval/compile.rs`, at compile time through `Host::convert_color` | the effective `colorspace` → working; none when no scope declares one | ✅ correct per MaterialX | | same, fed by an `image` | `crust-assets/src/uv_texture/` | the `file`'s effective `colorspace` → working, as for any texture | ✅ correct per MaterialX | [^1]: `baseColor`, `specularColor`, `transmissionColor`, `transmissionScatter`, diff --git a/docs/rust_leverage.md b/docs/rust_leverage.md index db6c49c9..db360fc9 100644 --- a/docs/rust_leverage.md +++ b/docs/rust_leverage.md @@ -310,7 +310,7 @@ is the first step of §4.6. | sentinel | location | better | |----------|----------|--------| | `HitRecord::NO_FACE = u32::MAX`, `face_uv` "meaningless" when set | `hittable.rs` | `Option` ties the uv to the id | -| `base_face = u32::MAX` | `scene/subdiv.rs`, `usd_import/mesh.rs` | same | +| `base_face = u32::MAX` | `scene/subdiv/`, `usd_import/mesh.rs` | same | | `indirect_clamp == 0.0` means off | `tracer/settings.rs` | `Option` validated at construction | | `STRIPE = usize::MAX` means unset | `tiled/cache.rs` | `Cell>` | diff --git a/docs/shading_performance.md b/docs/shading_performance.md index c2485d56..02335a68 100644 --- a/docs/shading_performance.md +++ b/docs/shading_performance.md @@ -24,7 +24,7 @@ the current state. Two materials evaluate a pattern network at every hit: - **`MtlxMaterial`** (`material/materialx.rs`). It ran the compiled - MaterialX `Program` (`crust-mtlx/src/eval.rs`), then `reduce` pooled the + MaterialX `Program` (`crust-mtlx/src/eval/`), then `reduce` pooled the document's lobes onto an `OpenPBR`. (That reduction has since been replaced by a closure tree collapsed into a `ResolvedClosure`; see "Where it stands now".) The `Program` was already a linear, slot-indexed instruction list with a diff --git a/openspec/specs/image-output/design.md b/openspec/specs/image-output/design.md index 9b8f9131..49ae4d40 100644 --- a/openspec/specs/image-output/design.md +++ b/openspec/specs/image-output/design.md @@ -11,8 +11,7 @@ the `aovs` capability's. beside it. This is the path every sample, golden and test without products takes, and it is kept exactly as it was before products existed, so their output did not move. -- **Products authored.** `mod products`, inline in `crust-render/src/main.rs` - (the CLI crate keeps one source file): one EXR per accepted +- **Products authored.** `crust-render/src/products.rs`: one EXR per accepted product, at its `productName` (or `-o` for the first), then the PNG from the first product's beauty, beside that product. diff --git a/openspec/specs/image-output/spec.md b/openspec/specs/image-output/spec.md index 20c9c884..e02fdc98 100644 --- a/openspec/specs/image-output/spec.md +++ b/openspec/specs/image-output/spec.md @@ -4,7 +4,8 @@ Persist the rendered pixel buffer to disk. The renderer writes a linear high-dynamic-range EXR, then converts it to a tone-mapped sRGB PNG for viewing. -Both steps live in `crust-render/src/main.rs`; the engine crate only produces +Both steps live in the `crust-render` CLI (`main.rs`, and `products.rs` for +authored `RenderProduct`s); the engine crate only produces the `Buffer`. ## Requirements diff --git a/openspec/specs/materials/design.md b/openspec/specs/materials/design.md index f5d7d12a..b4f9d29b 100644 --- a/openspec/specs/materials/design.md +++ b/openspec/specs/materials/design.md @@ -201,9 +201,10 @@ query then fails and the surface falls back to grey — which is what the two DPEL assets (MaterialX Teapot, MaterialX Lion) did on import. The reader is the standalone `crust-mtlx` crate — `parse.rs` (XML → a flat, - name-addressable graph), `value.rs` (the one runtime value), `eval.rs` (the - graph compiled to a slot-indexed program), `bsdf.rs` (the closure tree), - `surface.rs` (the three surface-shader nodes expanded into closure trees) — + name-addressable graph), `value.rs` (the one runtime value), `eval/` (the + graph compiled to a slot-indexed program: `compile.rs`, and the interpreter + in `apply.rs`), `bsdf.rs` (the closure tree), `surface/` (the three + surface-shader nodes expanded into closure trees, one file each) — and crust-core evaluates what it describes: `closure/` collapses the tree at a vertex and shades its leaves, `materialx.rs` is the `Material` and the importer-facing `load()`. @@ -560,7 +561,7 @@ - **Surface shaders are their nodegraphs, node for node.** `open_pbr_surface` (1.1), `standard_surface` (1.0.1) and `gltf_pbr` (2.0.1) are MaterialX nodedefs implemented as nodegraphs over standalone BSDFs, and a document - never carries the implementation. `surface.rs` reproduces each graph — the + never carries the implementation. `surface/` reproduces each graph — the same leaves, the same `layer` / `mix` / `multiply` in the same order, the same derived parameters emitted as program ops so they fold, optimise and JIT — commented with the nodegraph node names so it can be checked against the @@ -936,7 +937,7 @@ `normal`, `position` and `viewdirection` (the shading globals) are not in it, nor are `hsvadjust`, `colorcorrect` and `heighttonormal` (not yet in the script's `CATEGORIES`), nor anything the closure and surface builders - (`bsdf.rs`, `surface.rs`) emit, which the pattern ops they lower to are + (`bsdf.rs`, `surface/`) emit, which the pattern ops they lower to are checked through. `integer`, `boolean` and matrix signatures are not generated because crust has no such values. - **Stdlib nodegraphs lowered by hand.** `colorcorrect` (`color3`) is compiled to diff --git a/openspec/specs/usd-scene-import/design.md b/openspec/specs/usd-scene-import/design.md index 38837d13..3f4ae09b 100644 --- a/openspec/specs/usd-scene-import/design.md +++ b/openspec/specs/usd-scene-import/design.md @@ -141,7 +141,7 @@ Schema mapping: Non-invertible transforms still bake immediately, and a mesh authoring `crust:motion:translate` always instances (a baked mesh has no transform left to lerp). `UsdGeomSphere` → analytic `Sphere` geometry. -- **Subdivision surfaces** (`scene/subdiv.rs`, via the pure-Rust +- **Subdivision surfaces** (`scene/subdiv/`, via the pure-Rust [`opensubdiv-rs`](https://github.com/doubleailes/OpenSubdiv-rs) port of OpenSubdiv's Far/Sdc layers — zero dependencies, `forbid(unsafe_code)`, pinned to its `0.5.0` release tag, whose commit the committed `Cargo.lock` records). Read from USD, never from a crust From ff25d2f36182c0a78e6479d5960cbd235f332f29 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 5 Oct 2026 09:31:37 +0000 Subject: [PATCH 5/5] Bump the workspace version to 0.5.1 Also the docs site's landing-page version badge. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_019nwge6NCTPhuk1VPuZucRF --- Cargo.lock | 12 ++++++------ Cargo.toml | 2 +- site/content/_index.md | 2 +- 3 files changed, 8 insertions(+), 8 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 0e6b8265..04323f8f 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -528,7 +528,7 @@ checksum = "460fbee9c2c2f33933d720630a6a0bac33ba7053db5344fac858d4b8952d77d5" [[package]] name = "crust-assets" -version = "0.5.0" +version = "0.5.1" dependencies = [ "crust-core", "exr", @@ -542,7 +542,7 @@ dependencies = [ [[package]] name = "crust-core" -version = "0.5.0" +version = "0.5.1" dependencies = [ "criterion", "crust-jit", @@ -561,7 +561,7 @@ dependencies = [ [[package]] name = "crust-jit" -version = "0.5.0" +version = "0.5.1" dependencies = [ "cranelift-codegen", "cranelift-frontend", @@ -574,7 +574,7 @@ dependencies = [ [[package]] name = "crust-mtlx" -version = "0.5.0" +version = "0.5.1" dependencies = [ "glam", "roxmltree", @@ -582,7 +582,7 @@ dependencies = [ [[package]] name = "crust-render" -version = "0.5.0" +version = "0.5.1" dependencies = [ "clap", "crust-assets", @@ -599,7 +599,7 @@ dependencies = [ [[package]] name = "crust-rt" -version = "0.5.0" +version = "0.5.1" dependencies = [ "criterion", "glam", diff --git a/Cargo.toml b/Cargo.toml index d7754dcd..2e7e1fd1 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -12,7 +12,7 @@ edition = "2024" rust-version = "1.96" # SPDX form of LICENSE, so cargo-deny can check our own crates too. license = "MIT" -version = "0.5.0" +version = "0.5.1" # Nothing here goes to crates.io: crust-core and crust-assets depend on git # repositories, which crates.io refuses. Releases are the GitHub binaries. publish = false diff --git a/site/content/_index.md b/site/content/_index.md index 262815ba..523a7ba6 100644 --- a/site/content/_index.md +++ b/site/content/_index.md @@ -6,7 +6,7 @@ title = "A physically-based path tracer for USD" lead = 'Crust Render is a path tracer written in safe Rust. It renders USD scenes (.usda, .usdc, .usdz) with an OpenPBR übershader, MaterialX look-dev graphs, UsdLux lights, volumes and path guiding.' url = "/docs/getting-started/introduction/" url_button = "Get started" -repo_version = "GitHub v0.5.0" +repo_version = "GitHub v0.5.1" repo_license = "Open-source MIT License." repo_url = "https://github.com/doubleailes/crust-render"